pyterrier-tar 0.1.0__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- pyterrier_tar-0.1.0/BENCHMARKING.md +309 -0
- pyterrier_tar-0.1.0/CHANGELOG.md +143 -0
- pyterrier_tar-0.1.0/EVALUATION.md +29 -0
- pyterrier_tar-0.1.0/GAPS.md +37 -0
- pyterrier_tar-0.1.0/LICENSE.txt +373 -0
- pyterrier_tar-0.1.0/MANIFEST.in +10 -0
- pyterrier_tar-0.1.0/PKG-INFO +359 -0
- pyterrier_tar-0.1.0/PUBLISHED_EVIDENCE.md +75 -0
- pyterrier_tar-0.1.0/README.md +312 -0
- pyterrier_tar-0.1.0/RESULTS.md +104 -0
- pyterrier_tar-0.1.0/RESULTS_UNCERTAINTY.md +696 -0
- pyterrier_tar-0.1.0/THIRD_PARTY_NOTICES.md +16 -0
- pyterrier_tar-0.1.0/examples/benchmark_rankings.py +151 -0
- pyterrier_tar-0.1.0/examples/notebooks/demo_fresh_rankers_stopping.ipynb +364 -0
- pyterrier_tar-0.1.0/examples/notebooks/demo_multitopic_rankers_stopping.ipynb +404 -0
- pyterrier_tar-0.1.0/examples/notebooks/demo_publication_figures.ipynb +355 -0
- pyterrier_tar-0.1.0/examples/notebooks/demo_stopping_rules.ipynb +290 -0
- pyterrier_tar-0.1.0/examples/notebooks/demo_trajectory_rules.ipynb +415 -0
- pyterrier_tar-0.1.0/examples/notebooks/demo_workflows.ipynb +397 -0
- pyterrier_tar-0.1.0/examples/notebooks/parity_archives.ipynb +971 -0
- pyterrier_tar-0.1.0/examples/notebooks/parity_published.ipynb +223 -0
- pyterrier_tar-0.1.0/examples/notebooks.md +72 -0
- pyterrier_tar-0.1.0/examples/published_autostop_clef2017.py +106 -0
- pyterrier_tar-0.1.0/examples/published_baselines_clef.py +101 -0
- pyterrier_tar-0.1.0/examples/published_clef_tar_eval_metrics.py +160 -0
- pyterrier_tar-0.1.0/examples/published_cmh_buscar.py +131 -0
- pyterrier_tar-0.1.0/examples/published_ip_h_clef.py +125 -0
- pyterrier_tar-0.1.0/examples/published_kneedle_clef2017.py +65 -0
- pyterrier_tar-0.1.0/examples/published_point_process_clef2017.py +100 -0
- pyterrier_tar-0.1.0/examples/render_benchmark_figures.py +186 -0
- pyterrier_tar-0.1.0/examples/results_table.py +279 -0
- pyterrier_tar-0.1.0/pyproject.toml +73 -0
- pyterrier_tar-0.1.0/setup.cfg +4 -0
- pyterrier_tar-0.1.0/src/pyterrier_tar/__init__.py +51 -0
- pyterrier_tar-0.1.0/src/pyterrier_tar/datasets.py +251 -0
- pyterrier_tar-0.1.0/src/pyterrier_tar/experiments.py +359 -0
- pyterrier_tar-0.1.0/src/pyterrier_tar/grl.py +552 -0
- pyterrier_tar-0.1.0/src/pyterrier_tar/history.py +161 -0
- pyterrier_tar-0.1.0/src/pyterrier_tar/plotting.py +500 -0
- pyterrier_tar-0.1.0/src/pyterrier_tar/rankings.py +83 -0
- pyterrier_tar-0.1.0/src/pyterrier_tar/reporting.py +140 -0
- pyterrier_tar-0.1.0/src/pyterrier_tar/rules/__init__.py +26 -0
- pyterrier_tar-0.1.0/src/pyterrier_tar/rules/_base.py +262 -0
- pyterrier_tar-0.1.0/src/pyterrier_tar/rules/catalogue.py +81 -0
- pyterrier_tar-0.1.0/src/pyterrier_tar/rules/checkpoint/__init__.py +18 -0
- pyterrier_tar-0.1.0/src/pyterrier_tar/rules/checkpoint/_shared.py +18 -0
- pyterrier_tar-0.1.0/src/pyterrier_tar/rules/checkpoint/anytime_cmh.py +36 -0
- pyterrier_tar-0.1.0/src/pyterrier_tar/rules/checkpoint/cmh.py +68 -0
- pyterrier_tar-0.1.0/src/pyterrier_tar/rules/checkpoint/ip_hyperbolic.py +180 -0
- pyterrier_tar-0.1.0/src/pyterrier_tar/rules/checkpoint/poisson_point.py +126 -0
- pyterrier_tar-0.1.0/src/pyterrier_tar/rules/control/__init__.py +18 -0
- pyterrier_tar-0.1.0/src/pyterrier_tar/rules/control/_base.py +69 -0
- pyterrier_tar-0.1.0/src/pyterrier_tar/rules/control/baseline_inclusion_rate.py +62 -0
- pyterrier_tar-0.1.0/src/pyterrier_tar/rules/control/qbcb.py +91 -0
- pyterrier_tar-0.1.0/src/pyterrier_tar/rules/control/sampling.py +59 -0
- pyterrier_tar-0.1.0/src/pyterrier_tar/rules/control/target_recapture.py +48 -0
- pyterrier_tar-0.1.0/src/pyterrier_tar/rules/estimation/__init__.py +30 -0
- pyterrier_tar-0.1.0/src/pyterrier_tar/rules/estimation/autostop.py +95 -0
- pyterrier_tar-0.1.0/src/pyterrier_tar/rules/estimation/beta_binomial.py +59 -0
- pyterrier_tar-0.1.0/src/pyterrier_tar/rules/estimation/chao.py +49 -0
- pyterrier_tar-0.1.0/src/pyterrier_tar/rules/estimation/evpi.py +82 -0
- pyterrier_tar-0.1.0/src/pyterrier_tar/rules/estimation/quant.py +144 -0
- pyterrier_tar-0.1.0/src/pyterrier_tar/rules/estimation/scal.py +62 -0
- pyterrier_tar-0.1.0/src/pyterrier_tar/rules/metrics.py +218 -0
- pyterrier_tar-0.1.0/src/pyterrier_tar/rules/trajectory/__init__.py +28 -0
- pyterrier_tar-0.1.0/src/pyterrier_tar/rules/trajectory/_shared.py +39 -0
- pyterrier_tar-0.1.0/src/pyterrier_tar/rules/trajectory/batch_precision.py +82 -0
- pyterrier_tar-0.1.0/src/pyterrier_tar/rules/trajectory/budget.py +68 -0
- pyterrier_tar-0.1.0/src/pyterrier_tar/rules/trajectory/consecutive_irrelevant.py +48 -0
- pyterrier_tar-0.1.0/src/pyterrier_tar/rules/trajectory/fixed_round.py +44 -0
- pyterrier_tar-0.1.0/src/pyterrier_tar/rules/trajectory/kneedle.py +141 -0
- pyterrier_tar-0.1.0/src/pyterrier_tar/rules/trajectory/oracle.py +68 -0
- pyterrier_tar-0.1.0/src/pyterrier_tar/rules/trajectory/review_half.py +50 -0
- pyterrier_tar-0.1.0/src/pyterrier_tar/rules/trajectory/rule2399.py +50 -0
- pyterrier_tar-0.1.0/src/pyterrier_tar/rules/trajectory/safe.py +130 -0
- pyterrier_tar-0.1.0/src/pyterrier_tar.egg-info/PKG-INFO +359 -0
- pyterrier_tar-0.1.0/src/pyterrier_tar.egg-info/SOURCES.txt +89 -0
- pyterrier_tar-0.1.0/src/pyterrier_tar.egg-info/dependency_links.txt +1 -0
- pyterrier_tar-0.1.0/src/pyterrier_tar.egg-info/entry_points.txt +5 -0
- pyterrier_tar-0.1.0/src/pyterrier_tar.egg-info/requires.txt +23 -0
- pyterrier_tar-0.1.0/src/pyterrier_tar.egg-info/top_level.txt +1 -0
- pyterrier_tar-0.1.0/tests/test_compatibility_boundary.py +12 -0
- pyterrier_tar-0.1.0/tests/test_datasets.py +117 -0
- pyterrier_tar-0.1.0/tests/test_experiments.py +337 -0
- pyterrier_tar-0.1.0/tests/test_grl_persistence.py +113 -0
- pyterrier_tar-0.1.0/tests/test_history_rankings_plots.py +128 -0
- pyterrier_tar-0.1.0/tests/test_notebooks.py +79 -0
- pyterrier_tar-0.1.0/tests/test_package.py +9 -0
- pyterrier_tar-0.1.0/tests/test_publication_figures.py +267 -0
- pyterrier_tar-0.1.0/tests/test_replay_api.py +490 -0
- pyterrier_tar-0.1.0/tests/test_rule_validity.py +512 -0
|
@@ -0,0 +1,309 @@
|
|
|
1
|
+
# Reproducible replay benchmarks
|
|
2
|
+
|
|
3
|
+
The public experiment API runs recorded histories, saves per-topic results,
|
|
4
|
+
and reports uncertainty and failures. The input ranking is fixed before labels
|
|
5
|
+
are attached. This workflow does not train a ranker or select a stopping rule.
|
|
6
|
+
|
|
7
|
+
## Run and resume an experiment
|
|
8
|
+
|
|
9
|
+
```python
|
|
10
|
+
from pathlib import Path
|
|
11
|
+
import pyterrier_tar as tar
|
|
12
|
+
|
|
13
|
+
# trajectory has qid, docno, rank, label and covers the full collection.
|
|
14
|
+
data = tar.ReplayDataset("my-collection", "recorded-ranker", trajectory)
|
|
15
|
+
rules = [
|
|
16
|
+
tar.RuleSpec("IP-P", lambda c: tar.PoissonPoint(c.target, c.collection_size)),
|
|
17
|
+
tar.RuleSpec("50 irrelevant", lambda c: tar.ConsecutiveIrrelevant(50),
|
|
18
|
+
config={"count": 50}),
|
|
19
|
+
]
|
|
20
|
+
output = Path("artifacts/my-experiment")
|
|
21
|
+
results = tar.run_experiment([data], rules, targets=[.8, .9, .95],
|
|
22
|
+
seeds=[0], output=output)
|
|
23
|
+
summary = tar.experiment_summary(results)
|
|
24
|
+
summary.to_csv(output / "summary.csv", index=False)
|
|
25
|
+
(output / "REPORT.md").write_text(tar.experiment_report(results))
|
|
26
|
+
```
|
|
27
|
+
|
|
28
|
+
A factory receives `RuleContext(trajectory, target, seed, collection_size)`
|
|
29
|
+
and returns a fresh stopping rule. Built-in classifications are inferred from
|
|
30
|
+
the actual rule class using the catalogue, never from the display name:
|
|
31
|
+
|
|
32
|
+
```python
|
|
33
|
+
tar.RuleSpec("Oracle", lambda c: tar.Oracle(c.target))
|
|
34
|
+
tar.RuleSpec("IP-H", lambda c: tar.IPHyperbolic(c.target, c.collection_size))
|
|
35
|
+
```
|
|
36
|
+
|
|
37
|
+
Custom classes, including subclasses of built-in rules, require `promise=`:
|
|
38
|
+
`heuristic`, `estimator`, `certificate`, or `offline bound`. This is metadata,
|
|
39
|
+
not an automatically verified guarantee. Existing explicit classifications
|
|
40
|
+
remain supported, but a classification conflicting with a known built-in is
|
|
41
|
+
rejected. An absent `config` defaults to an empty dictionary.
|
|
42
|
+
|
|
43
|
+
Repetition is a separate setting: `repetitions="once"` uses the first supplied
|
|
44
|
+
seed; `repetitions="seeds"` uses every supplied seed. When omitted, certificates
|
|
45
|
+
default to `"seeds"` and other rules to `"once"`, preserving the earlier run
|
|
46
|
+
counts. For example, `tar.RuleSpec("QBCB", make_qbcb, repetitions="once")`
|
|
47
|
+
runs one control draw while retaining its certificate classification. A custom
|
|
48
|
+
stochastic estimator can use `promise="estimator", repetitions="seeds"`.
|
|
49
|
+
Factories must draw from `context.rng()`, not global randomness. It returns a
|
|
50
|
+
generator keyed by topic, target and seed, so draws are independent across all
|
|
51
|
+
three; `default_rng(context.seed)` alone repeats one stream for every topic and
|
|
52
|
+
target. The ranker is not part of the key, so compared rankings of one topic
|
|
53
|
+
share a draw. Varying ranker seeds
|
|
54
|
+
should be separate named rankings, not extra topics. Resolved `promise` and
|
|
55
|
+
`repetitions` are recorded on every result row; a factory cannot change its
|
|
56
|
+
classification between contexts. Completed tasks resume without rebuilding
|
|
57
|
+
their rule or resampling controls.
|
|
58
|
+
|
|
59
|
+
`manifest.json` binds the parameters, input-frame hashes, package source hash,
|
|
60
|
+
dependency versions and factory source (when available). An input hash covers
|
|
61
|
+
column names, row order and values, not how pandas stores them, so a run saved
|
|
62
|
+
under one pandas version or platform verifies under another. Include external
|
|
63
|
+
factory parameters and implementation versions in `RuleSpec.config`. Put
|
|
64
|
+
score snapshots and other frame inputs in `ReplayDataset.artifacts` and their
|
|
65
|
+
provenance in `metadata`; the manifest hashes those too. A factory closure is
|
|
66
|
+
not automatically a complete record of its external dependencies.
|
|
67
|
+
|
|
68
|
+
Completed tasks are atomically written under `tasks/`, with a result checksum.
|
|
69
|
+
`results.csv` is a regenerable export. Repeating the call resumes completed
|
|
70
|
+
tasks; changing the manifest requires a new directory. Failures leave completed
|
|
71
|
+
tasks intact. Only one writer may use a directory. After a process is killed,
|
|
72
|
+
verify that it has stopped before removing its `.writer.lock` directory.
|
|
73
|
+
Manifests use schema 3. A run saved with an earlier schema needs a new output
|
|
74
|
+
directory, and its inputs can no longer be verified for plot-only regeneration.
|
|
75
|
+
Existing result CSVs remain readable by the reporting and plotting APIs.
|
|
76
|
+
|
|
77
|
+
For a partial history, pass **full-pool qrels including negatives**:
|
|
78
|
+
|
|
79
|
+
```python
|
|
80
|
+
data = tar.ReplayDataset("collection", "partial-history", history.trajectory,
|
|
81
|
+
qrels=full_pool_qrels, control=history.control,
|
|
82
|
+
artifacts={"scores": history.score_snapshots,
|
|
83
|
+
"rounds": history.rounds},
|
|
84
|
+
metadata=history.metadata)
|
|
85
|
+
```
|
|
86
|
+
|
|
87
|
+
Omitting qrels declares the trajectory to be the entire collection. The runner
|
|
88
|
+
cannot infer unseen relevant documents. Full-pool qrels provide the denominator
|
|
89
|
+
and collection size; they do not supply missing review events to a rule. Rules
|
|
90
|
+
that need unobserved scores or sampling metadata still require those inputs.
|
|
91
|
+
|
|
92
|
+
The runner uses `stop_report` once and retains that prefix, avoiding a second
|
|
93
|
+
stochastic decision. Rules must follow the package's prefix-replay contract.
|
|
94
|
+
It obtains screened identities from `rule.control`, `rule.control_documents`,
|
|
95
|
+
and the dataset's baseline controls, then counts their union with the prefix.
|
|
96
|
+
Custom rules charging controls must expose their identities through these APIs.
|
|
97
|
+
|
|
98
|
+
## Read uncertainty and failures
|
|
99
|
+
|
|
100
|
+
`experiment_summary` supplies target attainment, Wilson 95% intervals, mean
|
|
101
|
+
review fraction with bootstrap intervals, minimum recall, mean/maximum recall
|
|
102
|
+
shortfall, and counts of actual firings, full-review fallbacks, exhausted partial
|
|
103
|
+
histories, and zero-relevant observations. `worst_topics` retains the topic,
|
|
104
|
+
target and seed of individual failures. `experiment_report` writes Markdown.
|
|
105
|
+
|
|
106
|
+
For certificate rules, each summary row is **one topic across independent control
|
|
107
|
+
draws**. It separates certification frequency from achieved-target frequency:
|
|
108
|
+
reviewing everything can reach the target without producing a certificate.
|
|
109
|
+
No pooled interval treats many draws on a few topics as many independent topics.
|
|
110
|
+
For once-only non-certificate rules, intervals use distinct topics, with
|
|
111
|
+
duplicates rejected. Explicitly repeated non-certificate rules are summarised
|
|
112
|
+
across seeds within each topic and labelled `seed`, not certificate coverage.
|
|
113
|
+
|
|
114
|
+
Review intervals use a seeded percentile bootstrap of the same observational
|
|
115
|
+
units; one observation produces no cost interval. Each summary row has its own
|
|
116
|
+
generator, so an interval is the same whether a rule is summarised alone or
|
|
117
|
+
alongside other rules and rankers. These descriptive intervals
|
|
118
|
+
do not adjust for selecting among methods or targets. Topic independence remains
|
|
119
|
+
an assumption. Increase the control draws to estimate coverage precisely; 20
|
|
120
|
+
draws per topic are only a coarse screen. Zero-relevant topics are counted and
|
|
121
|
+
retained with the metrics' recall convention of 1, so inspect their count.
|
|
122
|
+
|
|
123
|
+
## Compare rankings on the same pool
|
|
124
|
+
|
|
125
|
+
```bash
|
|
126
|
+
python examples/benchmark_rankings.py --years 2017 --limit 3 --waterloo \
|
|
127
|
+
--draws 100 --targets .9 --output artifacts/clef-development
|
|
128
|
+
```
|
|
129
|
+
|
|
130
|
+
This runs IP-P, ConsecutiveIrrelevant, ReviewHalf and QBCB on released AutoTAR,
|
|
131
|
+
BM25 using the topic title, and three seeded random rankings. It writes a
|
|
132
|
+
manifest, resumable tasks, results, summaries, failure rows, Markdown, and PNGs.
|
|
133
|
+
Remove `--limit` and omit `--years` to evaluate all three collections. A limited
|
|
134
|
+
run is a development check, not a new benchmark conclusion.
|
|
135
|
+
|
|
136
|
+
`--waterloo` adds the released Waterloo B ranking for CLEF 2017, pinned to the
|
|
137
|
+
same source revision and SHA-256 as the existing Kneedle parity check. This
|
|
138
|
+
supplies a second recorded review ranking; it is completed against the same
|
|
139
|
+
pool as AutoTAR. It is not a new training run or a reproduction of the original
|
|
140
|
+
ranking-generation procedure. The releases share 17 topic IDs; this option
|
|
141
|
+
restricts **all** rankers to those shared topics before applying `--limit`,
|
|
142
|
+
and records excluded AutoTAR topics in each dataset's manifest metadata.
|
|
143
|
+
|
|
144
|
+
`bm25_ranking` is a specified in-memory baseline: lowercase Unicode word tokens,
|
|
145
|
+
no stemming or stopword removal, unique query terms, k1=1.2, b=.75, and
|
|
146
|
+
`log(1 + (N-df+.5)/(df+.5))` IDF. It claims no Terrier scoring parity. Missing-text
|
|
147
|
+
documents remain in the pool with score zero. Random rankings discard labels
|
|
148
|
+
and are stable under input-row permutations. Both are generated before qrels
|
|
149
|
+
are joined. `complete_rankings(pool, ranking)` appends unranked documents by
|
|
150
|
+
docno and rejects foreign documents or absent topics.
|
|
151
|
+
|
|
152
|
+
To add another recorded learning trajectory, pass `--recorded histories.csv`.
|
|
153
|
+
The CSV needs `dataset,ranker,qid,docno,rank`, with dataset names such as
|
|
154
|
+
`clef2017` and topic-prefixed document IDs matching the CLEF adapter. Each named
|
|
155
|
+
ranker must cover every selected topic for its dataset. Missing documents are
|
|
156
|
+
appended, and the same full pool is used for every ranker. The package does not
|
|
157
|
+
invent an additional learned ranking if one is unavailable.
|
|
158
|
+
|
|
159
|
+
`examples/results_table.py` delegates to the public runner: `--checkpoint-dir`
|
|
160
|
+
enables resume, and it writes `RESULTS_UNCERTAINTY.md` beside `RESULTS.md` with
|
|
161
|
+
intervals and failures. `RESULTS.md` was regenerated this way. Its certificate
|
|
162
|
+
rows use control draws from `context.rng()`, so they differ by sampling noise
|
|
163
|
+
from the first table; every estimator and heuristic row is unchanged.
|
|
164
|
+
|
|
165
|
+
## Import review histories
|
|
166
|
+
|
|
167
|
+
Portable CSV events have `qid,docno,rank,label,round`. `rank` records review
|
|
168
|
+
order; `round` records batch membership, starting from zero or another
|
|
169
|
+
non-negative integer. Extra columns, including recorded sampling probabilities,
|
|
170
|
+
survive. Optional score CSVs have `qid,docno,index,score`, where `index` is the
|
|
171
|
+
number reviewed when the model produced that snapshot.
|
|
172
|
+
|
|
173
|
+
```python
|
|
174
|
+
history = tar.read_review_history("events.csv", scores="scores.csv")
|
|
175
|
+
# Or import existing DataFrames:
|
|
176
|
+
history = tar.import_review_history(events, score_snapshots=scores)
|
|
177
|
+
```
|
|
178
|
+
|
|
179
|
+
The TARexp adapter follows the pinned
|
|
180
|
+
[Ledger format](https://github.com/eugene-yang/tarexp/blob/d23724a73175cfacbaa95dbaf160a34bc17a1ac0/tarexp/ledger.py)
|
|
181
|
+
and [workflow score timing](https://github.com/eugene-yang/tarexp/blob/d23724a73175cfacbaa95dbaf160a34bc17a1ac0/tarexp/workflow.py).
|
|
182
|
+
For an already loaded ledger and saved score mapping:
|
|
183
|
+
|
|
184
|
+
```python
|
|
185
|
+
history = tar.import_tarexp(ledger, docnos=original_dataset_docnos,
|
|
186
|
+
qid="topic", saved_scores=saved_scores)
|
|
187
|
+
```
|
|
188
|
+
|
|
189
|
+
It also accepts the numeric N-by-2 ledger record array. Round -1 becomes a
|
|
190
|
+
separate control frame, round 0 is the seed batch, and NaN rows remain
|
|
191
|
+
unreviewed. Binary `predict_proba` arrays use column 1. Mapping keys are the
|
|
192
|
+
completed round numbers; snapshots become cumulative review indices.
|
|
193
|
+
`docnos` **must follow the original dataset row order**.
|
|
194
|
+
|
|
195
|
+
For a native TARexp `it_N` directory, with TARexp installed:
|
|
196
|
+
|
|
197
|
+
```python
|
|
198
|
+
history = tar.read_tarexp("saved_run/it_10", docnos=original_dataset_docnos,
|
|
199
|
+
qid="topic", trusted=True)
|
|
200
|
+
```
|
|
201
|
+
|
|
202
|
+
Native TARexp files are gzipped Python pickles and may execute code; only load
|
|
203
|
+
trusted checkpoints. The importer loads the ledger and workflow configuration.
|
|
204
|
+
Portable CSV import needs neither pickle nor TARexp installed.
|
|
205
|
+
|
|
206
|
+
TARexp records batch membership, not within-batch review order. Its imported
|
|
207
|
+
order is original dataset row order inside each batch. Document-level stopping
|
|
208
|
+
inside such a batch is not a reconstructed historical decision. For Quant,
|
|
209
|
+
use the preserved endpoints and model snapshots:
|
|
210
|
+
|
|
211
|
+
```python
|
|
212
|
+
rule = tar.QuantCI(.9, min_documents=1,
|
|
213
|
+
score_snapshots=history.score_snapshots,
|
|
214
|
+
review_checkpoints=history.rounds)
|
|
215
|
+
```
|
|
216
|
+
|
|
217
|
+
This replaces fixed-size checkpoint spacing, while retaining `min_documents`.
|
|
218
|
+
No future snapshot is used at an earlier checkpoint. Missing score rounds are
|
|
219
|
+
listed in metadata; Quant uses its documented latest-available-snapshot policy,
|
|
220
|
+
which is not per-round model parity when snapshots are missing. Scores remain
|
|
221
|
+
recorded scores: importing values in [0,1] does not establish calibration.
|
|
222
|
+
Quant's ranking must cover the full scored collection; a partial history or a
|
|
223
|
+
history with omitted control documents does not suffice for its denominator.
|
|
224
|
+
|
|
225
|
+
## Plot diagnostics
|
|
226
|
+
|
|
227
|
+
For an executable walkthrough with inline figures, open
|
|
228
|
+
[demo_publication_figures.ipynb](examples/notebooks/demo_publication_figures.ipynb).
|
|
229
|
+
It uses the saved development run, verifies its inputs, and exposes the dataset,
|
|
230
|
+
target, topic and export settings near the top.
|
|
231
|
+
For fresh rankings and stopping decisions across ten topics by default, use
|
|
232
|
+
[demo_multitopic_rankers_stopping.ipynb](examples/notebooks/demo_multitopic_rankers_stopping.ipynb).
|
|
233
|
+
Set `N_TOPICS` to another count or `None` for the full year. `PLOT_QID` and
|
|
234
|
+
`DIAGNOSTIC_RULE` select only the paired diagnostic; the overview uses all topics.
|
|
235
|
+
|
|
236
|
+
Install the optional `plot` extra. For publication figures, select one dataset
|
|
237
|
+
and target, then use the panel constructors and explicit export function:
|
|
238
|
+
|
|
239
|
+
```python
|
|
240
|
+
fig = tar.figure_recall_effort(topic_trajectory, topic_results, qrels=full_pool_qrels)
|
|
241
|
+
tar.save_figure(fig, "figures/recall_effort")
|
|
242
|
+
fig = tar.figure_target_misses(selected_results, n=15)
|
|
243
|
+
tar.save_figure(fig, "figures/target_misses")
|
|
244
|
+
fig = tar.figure_control_cost(selected_results)
|
|
245
|
+
tar.save_figure(fig, "figures/control_cost")
|
|
246
|
+
# All topics and methods for one ranker, dataset and target:
|
|
247
|
+
ranker_results = selected_results.loc[selected_results.ranker == "BM25"]
|
|
248
|
+
fig = tar.figure_stopping_outcomes(ranker_results)
|
|
249
|
+
tar.save_figure(fig, "figures/all_topics")
|
|
250
|
+
# One topic and method, separating ranked-prefix and total metrics:
|
|
251
|
+
fig = tar.figure_stopping_diagnostic(
|
|
252
|
+
topic_trajectory, topic_results.loc[topic_results.rule == "QBCB"], qrels=full_pool_qrels)
|
|
253
|
+
tar.save_figure(fig, "figures/qbcb_diagnostic")
|
|
254
|
+
```
|
|
255
|
+
|
|
256
|
+
Figures are 180 mm wide with 8–10 pt type, a white background, percentage axes,
|
|
257
|
+
shared panel scales, and legends in a separate strip. Cost components have
|
|
258
|
+
distinct colours and a hatch pattern. Each stopping rule has its own
|
|
259
|
+
combination of marker colour and shape: the benchmark's four rules keep fixed
|
|
260
|
+
ones, and any other rule takes one set by its name, so it looks the same
|
|
261
|
+
across figures. Crosses are reserved for non-firing terminations.
|
|
262
|
+
PDF exports embed TrueType fonts, SVG retains editable text, and PNG uses
|
|
263
|
+
600 dpi. `save_figure` takes an output stem and appends each format, so
|
|
264
|
+
`figures/cost_target0.9` exports `cost_target0.9.pdf`. These are journal-neutral defaults; journal-specific sizing can be
|
|
265
|
+
applied to the returned Figure. `save_figure` preserves the requested physical
|
|
266
|
+
width instead of cropping the canvas to a different size. Matplotlib's
|
|
267
|
+
[constrained layout](https://matplotlib.org/stable/users/explain/axes/constrainedlayout_guide.html)
|
|
268
|
+
allocates space for labels and panels; rendered layout tests check for clipped
|
|
269
|
+
or overlapping text. The lower-level `plot_*` functions still return axes and
|
|
270
|
+
accept an existing `ax=` for custom compositions.
|
|
271
|
+
|
|
272
|
+
Recall panels separate the stopping rules, so nearby outcomes cannot hide a
|
|
273
|
+
different method. The gain curve is a step function using offline labels. Stop
|
|
274
|
+
markers now use only prefix cost and prefix recall, so they lie on the curve.
|
|
275
|
+
Crosses mark non-firing terminations. Larger markers indicate coincident
|
|
276
|
+
outcomes without adding overlapping text; coordinates are never jittered.
|
|
277
|
+
For total cost and recall including controls, use `figure_stopping_outcomes`:
|
|
278
|
+
it shows every topic/seed outcome in a panel per method, with no ranking curve.
|
|
279
|
+
An open diamond shows one mean for each topic with repeated outcomes, regardless
|
|
280
|
+
of its draw count. Panel titles distinguish topics from outcomes; exact
|
|
281
|
+
coincidences may overlap. Means are not confidence intervals and a mean above
|
|
282
|
+
the target does not imply that all its draws succeeded. The paired diagnostic
|
|
283
|
+
puts ranked-prefix stops on the left and total outcomes on the right.
|
|
284
|
+
These visual explanations do not change
|
|
285
|
+
the stopping decisions or give a rule access to future labels.
|
|
286
|
+
|
|
287
|
+
The cost chart separates rankers, and stacks prefix cost and *additional*
|
|
288
|
+
controls to avoid double counting. It averages repeated draws within topics
|
|
289
|
+
before averaging across topics. Open circles show individual topic mean total
|
|
290
|
+
costs, not confidence intervals; slight vertical offsets keep topics distinct
|
|
291
|
+
without altering cost values. Failure panels select the largest individual
|
|
292
|
+
misses globally before grouping by ranker, retaining topic and control-draw
|
|
293
|
+
identity. Their caption reports the selection and shortfall definition.
|
|
294
|
+
|
|
295
|
+
Existing results can be re-rendered without running an experiment:
|
|
296
|
+
|
|
297
|
+
```bash
|
|
298
|
+
python examples/render_benchmark_figures.py --input artifacts/clef-validation
|
|
299
|
+
```
|
|
300
|
+
|
|
301
|
+
This refreshes the existing figures, adds an all-topic outcome figure per ranker
|
|
302
|
+
and a paired diagnostic, and writes PDF/SVG versions, `FIGURE_CAPTIONS.md`
|
|
303
|
+
and `FIGURE_PROVENANCE.json`. It verifies the stored result hash when available
|
|
304
|
+
and the reconstructed AutoTAR trajectory against the experiment manifest.
|
|
305
|
+
Numerical results, task checkpoints and the original manifest remain unchanged.
|
|
306
|
+
Other recall rankings can use `--trajectory` with the original full input CSV;
|
|
307
|
+
its hash must match the saved input. `--dataset`, `--target`, `--ranker`, `--qid`
|
|
308
|
+
and `--output` select the cohort and destination. The publication formatting
|
|
309
|
+
does not turn a limited development run into a general scientific result.
|
|
@@ -0,0 +1,143 @@
|
|
|
1
|
+
# Changelog
|
|
2
|
+
|
|
3
|
+
## 0.1.0
|
|
4
|
+
|
|
5
|
+
### Benchmark workflows
|
|
6
|
+
|
|
7
|
+
* `figure_stopping_outcomes` compares total screening effort and recall across
|
|
8
|
+
topics in separate method panels, showing individual outcomes and one mean
|
|
9
|
+
per repeated topic. `figure_stopping_diagnostic` separates ranked-prefix
|
|
10
|
+
stops from total outcomes in two panels. Recall-curve markers now use prefix
|
|
11
|
+
cost and recall, so they lie on the curve; total outcomes include controls
|
|
12
|
+
only in their separate scatter panels. Fresh-ranking notebooks expose both
|
|
13
|
+
views, with ten topics by default in the multi-topic example.
|
|
14
|
+
|
|
15
|
+
* `RuleSpec` infers built-in classifications from the catalogue and defaults
|
|
16
|
+
`config` to an empty dictionary. Custom classes require an explicit promise;
|
|
17
|
+
conflicting built-in classifications are rejected. `RuleContext.rng()` gives
|
|
18
|
+
each topic, target and seed its own generator. `repetitions="once"`
|
|
19
|
+
or `"seeds"` controls execution independently, with the previous repetition
|
|
20
|
+
counts as defaults. Repeated non-certificate methods are reported across
|
|
21
|
+
seeds within each topic. Manifests use schema 3, whose input hashes follow
|
|
22
|
+
values rather than pandas storage, so they verify across pandas versions and
|
|
23
|
+
accept array-valued columns such as `features`. Earlier manifests require a
|
|
24
|
+
fresh run directory; existing saved result tables can still be plotted and
|
|
25
|
+
summarised.
|
|
26
|
+
|
|
27
|
+
* Publication figures separate rankers and stopping rules into labelled panels,
|
|
28
|
+
retain exact data coordinates, and keep legends outside the data area. PDF,
|
|
29
|
+
editable SVG and 600-dpi PNG exports use a 180-mm print width. Plot-only
|
|
30
|
+
regeneration verifies saved inputs and preserves the numerical experiment.
|
|
31
|
+
Every rule has its own marker colour and shape, export stems may contain
|
|
32
|
+
dots, and outcome columns reloaded as 0/1 are accepted.
|
|
33
|
+
|
|
34
|
+
* Public `ReplayDataset`, `RuleSpec` and `run_experiment` APIs save per-topic
|
|
35
|
+
results, input/configuration hashes and atomic task checkpoints for resume.
|
|
36
|
+
Controls are counted once; exhausted partial histories remain distinct from
|
|
37
|
+
full-review fallbacks.
|
|
38
|
+
* `experiment_summary`, `experiment_report` and `worst_topics` expose intervals,
|
|
39
|
+
recall shortfalls and termination counts. Certificate intervals remain
|
|
40
|
+
within-topic across draws, with certification frequency reported separately.
|
|
41
|
+
Each summary row seeds its own bootstrap, so its review interval does not
|
|
42
|
+
depend on which other rules or rankers are summarised with it.
|
|
43
|
+
* Label-independent BM25/random baselines and completion of external rankings
|
|
44
|
+
support same-pool comparisons. `examples/benchmark_rankings.py` runs CLEF
|
|
45
|
+
comparisons and exports results and figures; `results_table.py` now uses the
|
|
46
|
+
shared runner and writes an uncertainty companion.
|
|
47
|
+
* Portable CSV and native TARexp history import preserve batches, controls,
|
|
48
|
+
score timing and missing-history boundaries. Quant/QuantCI accept explicit
|
|
49
|
+
recorded review checkpoints.
|
|
50
|
+
* Optional matplotlib diagnostics show gain curves/stops, target misses and
|
|
51
|
+
screening costs. Usage and evidence boundaries are in `BENCHMARKING.md`.
|
|
52
|
+
|
|
53
|
+
### Rules
|
|
54
|
+
|
|
55
|
+
* `CLEFTARDataset.complete_ranking` takes the pool to be every judged document
|
|
56
|
+
as well as every document with text. Ten CLEF topics judge documents the
|
|
57
|
+
archive has no text for (2,384 in CD009263, two of them relevant); it
|
|
58
|
+
rejected the released AutoTAR ranking for those topics, and would have left
|
|
59
|
+
such documents out of a completed ranking.
|
|
60
|
+
* Every rule with a recall target takes `target_recall` first and requires
|
|
61
|
+
it. `IPHyperbolic` and `PoissonPoint` take `(target_recall, collection_size)`
|
|
62
|
+
like `CMHHeuristic`; `QBCB` and `TargetRecapture` no longer default the
|
|
63
|
+
target; `BaselineInclusionRate` takes `(baseline, target_recall,
|
|
64
|
+
collection_size)`. `Quant` and the EVPI variants list their real parameters.
|
|
65
|
+
* `AutoStop` takes `collection_size`, which `strict_v2` now requires. It
|
|
66
|
+
previously used the sample size as the population, so its finite-population
|
|
67
|
+
correction shrank the variance towards zero and it stopped too early; at the
|
|
68
|
+
final checkpoint it equalled `loose`. `published_autostop_clef2017.py` now
|
|
69
|
+
checks the rule's public replay rather than its internal estimator.
|
|
70
|
+
* Every match of a query across frames compares `qid` as text: qrels in
|
|
71
|
+
`stopping_metrics`, control sets, `QuantCI` snapshots, `SAFE`'s key papers
|
|
72
|
+
and seed counts, and `review_fraction`'s size mapping. A ranking keyed `1`
|
|
73
|
+
against qrels keyed `'1'` previously scored recall 1.0 on every topic, and a
|
|
74
|
+
control rule reported no controls and zero control cost. `stopping_metrics`
|
|
75
|
+
now raises when no ranked query appears in the qrels at all.
|
|
76
|
+
* `SAFE` needs a `docno` column only when it has key papers.
|
|
77
|
+
* `BetaBinomial`, `SAFE`, `stopping_metrics`, and `stopping_frontier` take an
|
|
78
|
+
optional `collection_size` (per query for `SAFE` and the metrics). Without it
|
|
79
|
+
they treat the trajectory as the whole collection, as before, which on a
|
|
80
|
+
truncated ranking understates what remains, lowers SAFE's minimum depth, and
|
|
81
|
+
inflates `review_fraction` and `loss_e`. The README states the contract.
|
|
82
|
+
* `QBCB` and `TargetRecapture` say when a positive control is missing from the
|
|
83
|
+
ranking, instead of reporting only that the control set cannot certify.
|
|
84
|
+
* Trajectories with a missing label are rejected instead of reading it as
|
|
85
|
+
irrelevant, and `Chao` rejects fractional capture counts instead of
|
|
86
|
+
truncating them. `Chao`'s documentation now says it uses classic Chao1,
|
|
87
|
+
bias-corrected only when `f2 = 0`.
|
|
88
|
+
* `Kneedle` and `Budget` take `knee_distance`, defaulting to `'absolute'`: the
|
|
89
|
+
form every released TAR implementation uses, including the one behind the
|
|
90
|
+
86,243-effort figure. `'signed'` keeps Satopää et al.'s Kneedle. The two agree
|
|
91
|
+
on that run but diverge sharply on the CLEF AutoTAR rankings.
|
|
92
|
+
* `IPHyperbolic` reproduces the released loop's empty-tail stop and its
|
|
93
|
+
dynamic relevant-document gate, so all nine CLEF 2017--19 settings now match
|
|
94
|
+
the published recall, cost, reliability, loss, and relative error. `tail`
|
|
95
|
+
selects the released expected-remaining formula (the default, which
|
|
96
|
+
understates the tail by `(1-b)^2`) or the integral of the fitted rate.
|
|
97
|
+
* `SCAL` selects S-CAL's cutoff from the Horvitz-Thompson total over the whole
|
|
98
|
+
recorded sample, and charges the sample after the cutoff as review cost.
|
|
99
|
+
`control_documents()` lets `stopping_metrics` count it towards recall too.
|
|
100
|
+
It previously fired at the first checkpoint of a fully reviewed stratum.
|
|
101
|
+
* `BaselineInclusionRate` no longer reports a fired stop at position 0 when its
|
|
102
|
+
pilot sample holds no relevant documents.
|
|
103
|
+
* `BatchPrecision` counts a batch at the cutoff as low precision, as TARexp does.
|
|
104
|
+
* `QuantCI` accepts recorded per-round `score_snapshots`, and raises when a
|
|
105
|
+
query is missing from them rather than reporting it as never certified.
|
|
106
|
+
* `GRLStop.save` writes a configuration sidecar and stages both files before
|
|
107
|
+
publishing either; `load` restores the trained configuration, keeps this
|
|
108
|
+
instance's `target_recall` and `deterministic`, warns outside the trained
|
|
109
|
+
targets, and refuses metadata whose recorded digest does not match the policy.
|
|
110
|
+
`early_stopping` selects the released callback or a rollout-mean variant.
|
|
111
|
+
* Round-off fixes: `ceil(.07 * 100)` no longer inflates a recall target, and
|
|
112
|
+
`BetaBinomial` no longer loses an allowed miss to floating point.
|
|
113
|
+
|
|
114
|
+
### Usability
|
|
115
|
+
|
|
116
|
+
* `catalogue()` lists every rule with its family, what it promises, what it
|
|
117
|
+
needs beyond a labelled trajectory, and its source; a test keeps it in step
|
|
118
|
+
with the exports.
|
|
119
|
+
* A non-numeric `label` column now explains what a label should be instead of
|
|
120
|
+
surfacing pandas' parse error.
|
|
121
|
+
|
|
122
|
+
### Evaluation
|
|
123
|
+
|
|
124
|
+
* `RESULTS.md`, written by `examples/results_table.py`, reports every rule
|
|
125
|
+
that can run on the CLEF release: reliability and review at targets 0.8,
|
|
126
|
+
0.9, and 0.95 per collection, grouped by promise and not ranked.
|
|
127
|
+
`RESULTS_UNCERTAINTY.md` gives the intervals and largest shortfalls.
|
|
128
|
+
* `reliability_summary` and `coverage_summary`, because the families promise
|
|
129
|
+
different things: an estimator's promise is across topics, a certificate's is
|
|
130
|
+
across repeated control draws. `EVALUATION.md` says which to use for each.
|
|
131
|
+
* `published_kneedle_clef2017.py`, `published_point_process_clef2017.py`,
|
|
132
|
+
`published_autostop_clef2017.py`, and `published_baselines_clef.py` join the
|
|
133
|
+
parity scripts; all seven run weekly in CI and from `parity_published.ipynb`.
|
|
134
|
+
|
|
135
|
+
### Layout
|
|
136
|
+
|
|
137
|
+
* Each rule has its own file, grouped by what it reads: `rules/trajectory/`,
|
|
138
|
+
`rules/checkpoint/`, `rules/control/`, `rules/estimation/`. README documents
|
|
139
|
+
how to add one.
|
|
140
|
+
* Five notebooks, each either a `demo_` or a `parity_`.
|
|
141
|
+
* `GAPS.md` collects what the package does not implement or reproduce;
|
|
142
|
+
`EVALUATION.md` how to measure each family; `PUBLISHED_EVIDENCE.md` the parity claims and the
|
|
143
|
+
dated re-verification log.
|
|
@@ -0,0 +1,29 @@
|
|
|
1
|
+
# How to evaluate a stopping rule
|
|
2
|
+
|
|
3
|
+
Stopping rules in this package promise three different things, and a single
|
|
4
|
+
leaderboard hides that. What a rule promises decides what would falsify it.
|
|
5
|
+
|
|
6
|
+
| Family | Rules | Promise | The check |
|
|
7
|
+
| --- | --- | --- | --- |
|
|
8
|
+
| Heuristic | Kneedle, Budget, Rule2399, ReviewHalf, FixedRound, BatchPrecision, ConsecutiveIrrelevant, SAFE | none about recall: they detect a flattening gain curve or exhaust a budget | the cost/recall trade-off, and excess cost over the oracle depth |
|
|
9
|
+
| Estimator | CMH, AnytimeCMH, IP-P, IP-H, Quant/QuantCI, AutoStop, SCAL, Chao, BetaBinomial | recall at or above the target **if its model holds** | `reliability_summary`: the share of topics reaching the target, with a Wilson interval, against the nominal level |
|
|
10
|
+
| Certificate | QBCB, TargetRecapture | recall at or above the target with probability `1 - alpha` **over the draw of the control sample** | `coverage_summary`: the share of repeated draws that certified the target, for one review |
|
|
11
|
+
|
|
12
|
+
The unit of repetition is the difference that matters. An estimator's promise
|
|
13
|
+
is about topics, so it is measured across them. A certificate's promise is
|
|
14
|
+
about the sample it drew, so it is measured by drawing again: coverage across
|
|
15
|
+
replications of one review. Measuring a certificate across topics answers a
|
|
16
|
+
question nobody asked.
|
|
17
|
+
|
|
18
|
+
Costs are comparable only when the control screening is charged. `stop_report`
|
|
19
|
+
and `stopping_metrics` already count it once; a certificate that costs more
|
|
20
|
+
than a heuristic is buying a guarantee the heuristic never offers.
|
|
21
|
+
|
|
22
|
+
The public `run_experiment` workflow applies these distinctions to saved runs.
|
|
23
|
+
`experiment_summary` reports topic intervals for non-certificates and per-topic
|
|
24
|
+
draw intervals for certificates, alongside shortfalls and non-firing terminations.
|
|
25
|
+
`experiment_report` writes the tables and largest failures. See
|
|
26
|
+
[BENCHMARKING.md](BENCHMARKING.md) for resume, input provenance, ranker comparisons,
|
|
27
|
+
history import and plotting. Intervals are descriptive; full-review successes
|
|
28
|
+
are counted separately from certifications, and zero-relevant observations are
|
|
29
|
+
explicitly identified.
|
|
@@ -0,0 +1,37 @@
|
|
|
1
|
+
# Known gaps
|
|
2
|
+
|
|
3
|
+
What this package does not implement or cannot reproduce, and why. Parity
|
|
4
|
+
claims live in [PUBLISHED_EVIDENCE.md](PUBLISHED_EVIDENCE.md); this file is the
|
|
5
|
+
list of what is missing.
|
|
6
|
+
|
|
7
|
+
## Published results that cannot be reproduced
|
|
8
|
+
|
|
9
|
+
These need data the authors did not release. Nothing in the package should
|
|
10
|
+
claim them.
|
|
11
|
+
|
|
12
|
+
| Result | Missing input |
|
|
13
|
+
| --- | --- |
|
|
14
|
+
| QBCB's RCV1 figures | the 20% subset, seeds, active-learning trajectories, and control-sample identities |
|
|
15
|
+
| Yang et al. (2021) heuristics on RCV1 | the saved review histories |
|
|
16
|
+
| Adaptive AutoStop | the per-draw sampling distributions, which a trajectory cannot reconstruct |
|
|
17
|
+
| Chao et al. Table 7 integer displays | the analysis and formatting code; raw aggregates do not round to the printed integers (57.566 prints as 57) |
|
|
18
|
+
| The `Knee` rows of the point-process baseline table | the 2018 qrels are rebuilt in that repository from a PID list it does not ship; `published_baselines_clef.py` reports the difference instead of asserting it |
|
|
19
|
+
| `TM` and `TM-adapted` rows of the same table | the released target method draws its target set with an unseeded `random.choice` |
|
|
20
|
+
| RLStop and GRLStop paper results | no trained policies or result tables were released |
|
|
21
|
+
|
|
22
|
+
## Methods named but not implemented
|
|
23
|
+
|
|
24
|
+
| Name | State |
|
|
25
|
+
| --- | --- |
|
|
26
|
+
| Score-distribution stopping (Hollmann and Eickhoff) | not implemented; `QuantCI` over calibrated probabilities is the nearest rule |
|
|
27
|
+
| The adapted target method of the point-process papers | not implemented; `TargetRecapture` is Cormack and Grossman's original target method |
|
|
28
|
+
| TARexp-style Knee and Budget | TARexp takes the maximum slope ratio over every round split, measured in rounds. This package follows Cormack and Grossman's knee. An opt-in mode would be needed for TARexp-equivalent numbers. |
|
|
29
|
+
| Adaptive AutoStop, a live S-CAL controller | out of scope: this package replays recorded trajectories and does not select documents |
|
|
30
|
+
|
|
31
|
+
## Engineering and release
|
|
32
|
+
|
|
33
|
+
| Gap | Notes |
|
|
34
|
+
| --- | --- |
|
|
35
|
+
| CLEF archives come from one mirror | `npai.science.uu.nl`, hash-pinned. The hashes protect the content, not its availability |
|
|
36
|
+
| 9 of 119 SYNERGY datasets will not compose | missing source files or server errors at the pinned commit |
|
|
37
|
+
| Parity runs weekly, archive audits by hand | the `published-parity` job downloads about 1 GB; the archive notebooks need artifacts you supply |
|