croissantminer 0.2.0__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- croissantminer-0.2.0/LICENSE +21 -0
- croissantminer-0.2.0/PKG-INFO +379 -0
- croissantminer-0.2.0/README.md +330 -0
- croissantminer-0.2.0/croissantminer/__init__.py +41 -0
- croissantminer-0.2.0/croissantminer/__main__.py +5 -0
- croissantminer-0.2.0/croissantminer/api.py +106 -0
- croissantminer-0.2.0/croissantminer/cli.py +198 -0
- croissantminer-0.2.0/croissantminer/confidence.py +201 -0
- croissantminer-0.2.0/croissantminer/config.py +153 -0
- croissantminer-0.2.0/croissantminer/croissant.py +240 -0
- croissantminer-0.2.0/croissantminer/evaluator.py +449 -0
- croissantminer-0.2.0/croissantminer/extractor.py +293 -0
- croissantminer-0.2.0/croissantminer/field_filter.py +120 -0
- croissantminer-0.2.0/croissantminer/field_types.py +278 -0
- croissantminer-0.2.0/croissantminer/groundtruth_parser.py +72 -0
- croissantminer-0.2.0/croissantminer/groundtruth_parser_md.py +222 -0
- croissantminer-0.2.0/croissantminer/llm_evaluator.py +546 -0
- croissantminer-0.2.0/croissantminer/markdown_generator.py +288 -0
- croissantminer-0.2.0/croissantminer/methods.py +324 -0
- croissantminer-0.2.0/croissantminer/metrics.py +911 -0
- croissantminer-0.2.0/croissantminer/models/__init__.py +5 -0
- croissantminer-0.2.0/croissantminer/models/base.py +18 -0
- croissantminer-0.2.0/croissantminer/models/claude_model.py +321 -0
- croissantminer-0.2.0/croissantminer/models/factory.py +84 -0
- croissantminer-0.2.0/croissantminer/models/gemini_model.py +145 -0
- croissantminer-0.2.0/croissantminer/models/openai_model.py +62 -0
- croissantminer-0.2.0/croissantminer/models/qwen_model.py +279 -0
- croissantminer-0.2.0/croissantminer/pdf/__init__.py +22 -0
- croissantminer-0.2.0/croissantminer/pdf/processor.py +1115 -0
- croissantminer-0.2.0/croissantminer/pdf/reader.py +71 -0
- croissantminer-0.2.0/croissantminer/react_agent/__init__.py +7 -0
- croissantminer-0.2.0/croissantminer/react_agent/agent.py +761 -0
- croissantminer-0.2.0/croissantminer/react_agent/audit.py +138 -0
- croissantminer-0.2.0/croissantminer/react_agent/runner.py +231 -0
- croissantminer-0.2.0/croissantminer/react_agent/schemas.py +72 -0
- croissantminer-0.2.0/croissantminer/react_agent/tools.py +395 -0
- croissantminer-0.2.0/croissantminer/relevance.py +108 -0
- croissantminer-0.2.0/croissantminer/reporter.py +490 -0
- croissantminer-0.2.0/croissantminer/systems/__init__.py +12 -0
- croissantminer-0.2.0/croissantminer/systems/helpers.py +452 -0
- croissantminer-0.2.0/croissantminer/systems/locator_extractor.py +1233 -0
- croissantminer-0.2.0/croissantminer/systems/sections.py +588 -0
- croissantminer-0.2.0/croissantminer/systems/specialist_prompts.py +326 -0
- croissantminer-0.2.0/croissantminer/systems/specialists.py +305 -0
- croissantminer-0.2.0/croissantminer/systems/triage_critique.py +967 -0
- croissantminer-0.2.0/croissantminer/systems/validate.py +90 -0
- croissantminer-0.2.0/croissantminer/unifier.py +227 -0
- croissantminer-0.2.0/croissantminer.egg-info/PKG-INFO +379 -0
- croissantminer-0.2.0/croissantminer.egg-info/SOURCES.txt +59 -0
- croissantminer-0.2.0/croissantminer.egg-info/dependency_links.txt +1 -0
- croissantminer-0.2.0/croissantminer.egg-info/entry_points.txt +2 -0
- croissantminer-0.2.0/croissantminer.egg-info/requires.txt +27 -0
- croissantminer-0.2.0/croissantminer.egg-info/top_level.txt +1 -0
- croissantminer-0.2.0/pyproject.toml +77 -0
- croissantminer-0.2.0/setup.cfg +4 -0
- croissantminer-0.2.0/tests/test_cli.py +105 -0
- croissantminer-0.2.0/tests/test_croissant_output.py +108 -0
- croissantminer-0.2.0/tests/test_extraction_output.py +83 -0
- croissantminer-0.2.0/tests/test_leaderboard.py +54 -0
- croissantminer-0.2.0/tests/test_scoring_rules.py +84 -0
- croissantminer-0.2.0/tests/test_table2_reproduction.py +74 -0
|
@@ -0,0 +1,21 @@
|
|
|
1
|
+
MIT License
|
|
2
|
+
|
|
3
|
+
Copyright (c) 2026 The CroissantMiner Authors
|
|
4
|
+
|
|
5
|
+
Permission is hereby granted, free of charge, to any person obtaining a copy
|
|
6
|
+
of this software and associated documentation files (the "Software"), to deal
|
|
7
|
+
in the Software without restriction, including without limitation the rights
|
|
8
|
+
to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
|
|
9
|
+
copies of the Software, and to permit persons to whom the Software is
|
|
10
|
+
furnished to do so, subject to the following conditions:
|
|
11
|
+
|
|
12
|
+
The above copyright notice and this permission notice shall be included in all
|
|
13
|
+
copies or substantial portions of the Software.
|
|
14
|
+
|
|
15
|
+
THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
|
|
16
|
+
IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
|
|
17
|
+
FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
|
|
18
|
+
AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
|
|
19
|
+
LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
|
|
20
|
+
OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
|
|
21
|
+
SOFTWARE.
|
|
@@ -0,0 +1,379 @@
|
|
|
1
|
+
Metadata-Version: 2.4
|
|
2
|
+
Name: croissantminer
|
|
3
|
+
Version: 0.2.0
|
|
4
|
+
Summary: Extract Croissant metadata, including the Responsible AI fields, from ML dataset papers
|
|
5
|
+
Author-email: Berke Arda <bearda@ethz.ch>
|
|
6
|
+
License: MIT
|
|
7
|
+
Project-URL: Homepage, https://github.com/berkearda/croissantminer
|
|
8
|
+
Project-URL: Demo, https://huggingface.co/spaces/bearda/croissantminer
|
|
9
|
+
Project-URL: Dataset, https://huggingface.co/datasets/bearda/croissantminer
|
|
10
|
+
Project-URL: Issues, https://github.com/berkearda/croissantminer/issues
|
|
11
|
+
Project-URL: Changelog, https://github.com/berkearda/croissantminer/blob/main/CHANGELOG.md
|
|
12
|
+
Keywords: machine-learning,metadata,croissant,responsible-ai,datasets,llm,documentation
|
|
13
|
+
Classifier: Development Status :: 4 - Beta
|
|
14
|
+
Classifier: Intended Audience :: Science/Research
|
|
15
|
+
Classifier: License :: OSI Approved :: MIT License
|
|
16
|
+
Classifier: Programming Language :: Python :: 3
|
|
17
|
+
Classifier: Programming Language :: Python :: 3.10
|
|
18
|
+
Classifier: Programming Language :: Python :: 3.11
|
|
19
|
+
Classifier: Programming Language :: Python :: 3.12
|
|
20
|
+
Classifier: Programming Language :: Python :: 3.13
|
|
21
|
+
Classifier: Topic :: Scientific/Engineering :: Artificial Intelligence
|
|
22
|
+
Requires-Python: >=3.10
|
|
23
|
+
Description-Content-Type: text/markdown
|
|
24
|
+
License-File: LICENSE
|
|
25
|
+
Requires-Dist: anthropic<1,>=0.72.0
|
|
26
|
+
Requires-Dist: openai<2,>=1.76.0
|
|
27
|
+
Requires-Dist: PyPDF2>=3.0.1
|
|
28
|
+
Requires-Dist: requests
|
|
29
|
+
Requires-Dist: tqdm
|
|
30
|
+
Requires-Dist: python-dotenv
|
|
31
|
+
Provides-Extra: validate
|
|
32
|
+
Requires-Dist: mlcroissant<2,>=1.0; extra == "validate"
|
|
33
|
+
Provides-Extra: eval
|
|
34
|
+
Requires-Dist: pandas>=2.0.3; extra == "eval"
|
|
35
|
+
Requires-Dist: pyarrow>=14.0; extra == "eval"
|
|
36
|
+
Requires-Dist: numpy>=1.24.4; extra == "eval"
|
|
37
|
+
Requires-Dist: scipy>=1.10.1; extra == "eval"
|
|
38
|
+
Requires-Dist: scikit-learn>=1.3; extra == "eval"
|
|
39
|
+
Requires-Dist: matplotlib>=3.7; extra == "eval"
|
|
40
|
+
Requires-Dist: openpyxl>=3.1.0; extra == "eval"
|
|
41
|
+
Requires-Dist: regex; extra == "eval"
|
|
42
|
+
Requires-Dist: PyMuPDF>=1.24.0; extra == "eval"
|
|
43
|
+
Provides-Extra: demo
|
|
44
|
+
Requires-Dist: gradio<7,>=6.14; extra == "demo"
|
|
45
|
+
Provides-Extra: dev
|
|
46
|
+
Requires-Dist: pytest>=8.0; extra == "dev"
|
|
47
|
+
Requires-Dist: ruff>=0.5.0; extra == "dev"
|
|
48
|
+
Dynamic: license-file
|
|
49
|
+
|
|
50
|
+
<p align="center"><img src="https://raw.githubusercontent.com/berkearda/croissantminer/main/assets/croissantminer-logo.svg" width="520" alt="CroissantMiner"></p>
|
|
51
|
+
|
|
52
|
+
<p align="center">
|
|
53
|
+
Extract <a href="https://github.com/mlcommons/croissant">Croissant</a> metadata, including the 20 Responsible AI fields,
|
|
54
|
+
from the paper that introduces an ML dataset.
|
|
55
|
+
</p>
|
|
56
|
+
|
|
57
|
+
<p align="center">
|
|
58
|
+
<a href="https://huggingface.co/spaces/bearda/croissantminer"><img alt="Demo" src="https://img.shields.io/badge/demo-Hugging%20Face%20Space-ffcc4d"></a>
|
|
59
|
+
<a href="https://huggingface.co/datasets/bearda/croissantminer"><img alt="Dataset" src="https://img.shields.io/badge/dataset-Hugging%20Face-ffcc4d"></a>
|
|
60
|
+
<a href="https://github.com/berkearda/croissantminer/actions/workflows/tests.yml"><img alt="Tests" src="https://github.com/berkearda/croissantminer/actions/workflows/tests.yml/badge.svg"></a>
|
|
61
|
+
<img alt="Python 3.10 to 3.13" src="https://img.shields.io/badge/python-3.10%20to%203.13-3776ab">
|
|
62
|
+
<a href="https://github.com/berkearda/croissantminer/blob/main/LICENSE"><img alt="License: MIT" src="https://img.shields.io/badge/license-MIT-2ea44f"></a>
|
|
63
|
+
</p>
|
|
64
|
+
|
|
65
|
+
Code, benchmark and systems of **CroissantMiner: Automated Extraction and Validation of Croissant Metadata for ML
|
|
66
|
+
Datasets** (NeurIPS 2026, Evaluations and Datasets Track). Paper: arXiv link follows.
|
|
67
|
+
|
|
68
|
+
**News**
|
|
69
|
+
- **1 Oct 2026:** on PyPI (`pip install croissantminer`), with `--merge-into` to add the fields to a dataset's
|
|
70
|
+
Croissant file on Hugging Face, a
|
|
71
|
+
[guide for NeurIPS dataset submissions](https://github.com/berkearda/croissantminer/blob/main/docs/neurips.md) and a
|
|
72
|
+
[leaderboard](https://github.com/berkearda/croissantminer/blob/main/leaderboard/README.md) open to new systems.
|
|
73
|
+
- **30 Sep 2026:** `croissantminer extract`, one command from a paper to a Croissant file.
|
|
74
|
+
- **28 Sep 2026:** code, [dataset](https://huggingface.co/datasets/bearda/croissantminer) and
|
|
75
|
+
[demo](https://huggingface.co/spaces/bearda/croissantminer) released.
|
|
76
|
+
- **24 Sep 2026:** accepted at NeurIPS 2026 (Evaluations and Datasets Track).
|
|
77
|
+
|
|
78
|
+
Documenting a dataset in the [Croissant](https://github.com/mlcommons/croissant) format, including its Responsible
|
|
79
|
+
AI (RAI) fields, takes time, and venues such as the NeurIPS Evaluations and Datasets Track ask for it.
|
|
80
|
+
CroissantMiner reads the paper that introduces a dataset and drafts all 30 fields of the Croissant 1.1 schema: 10
|
|
81
|
+
core fields (name, license, creators and others) and 20 RAI fields (how the data was collected and annotated,
|
|
82
|
+
known biases, limitations, intended uses and others). You check the draft and publish it.
|
|
83
|
+
|
|
84
|
+
- **For dataset authors:** one command turns a paper into a Croissant file that passes the MLCommons validator.
|
|
85
|
+
- **For researchers:** a benchmark of 602 dataset papers with human gold annotations for 102 of them, the outputs
|
|
86
|
+
and scores of 24 extraction systems, and a [leaderboard](https://github.com/berkearda/croissantminer/blob/main/leaderboard/README.md) that scores new ones with the
|
|
87
|
+
paper's scorer.
|
|
88
|
+
|
|
89
|
+
## Try it in your browser
|
|
90
|
+
|
|
91
|
+
The [demo](https://huggingface.co/spaces/bearda/croissantminer) runs the six systems below on a PDF you upload.
|
|
92
|
+
It needs your own Anthropic or OpenAI API key, which is sent only to that provider and not stored.
|
|
93
|
+
|
|
94
|
+
<p align="center"><img src="https://raw.githubusercontent.com/berkearda/croissantminer/main/docs/figures/demo.png" width="760" alt="The demo after extracting the GSM8K paper with Triage + Critique: 19 of 30 fields, each with a supporting quote from the paper"></p>
|
|
95
|
+
|
|
96
|
+
## Quick start
|
|
97
|
+
|
|
98
|
+
Python 3.10 to 3.13.
|
|
99
|
+
|
|
100
|
+
```bash
|
|
101
|
+
pip install "croissantminer[validate]" # the extraction tool and the Croissant validator
|
|
102
|
+
export ANTHROPIC_API_KEY=... # or put it in a .env file in the folder you run from
|
|
103
|
+
croissantminer extract paper.pdf
|
|
104
|
+
```
|
|
105
|
+
|
|
106
|
+
```text
|
|
107
|
+
Extracting with Single-pass · Claude Sonnet 4.6 (usually about 30 s)...
|
|
108
|
+
Found 22 of 30 fields (core 9 of 10, Responsible AI 13 of 20) with single-pass in 35 s, about $0.06.
|
|
109
|
+
Not found: license, rai:dataCollectionMissingData, rai:dataCollectionTimeframe, ...
|
|
110
|
+
Wrote: paper.croissant.json (Croissant 1.1)
|
|
111
|
+
Check: passes the mlcroissant validator (2 recommended properties missing)
|
|
112
|
+
These are drafts by a language model: check each value against the paper before publishing.
|
|
113
|
+
```
|
|
114
|
+
|
|
115
|
+
<details>
|
|
116
|
+
<summary>Example output for GSM8K (<code>--hf-id openai/gsm8k</code>), shortened</summary>
|
|
117
|
+
|
|
118
|
+
```jsonc
|
|
119
|
+
{
|
|
120
|
+
"@context": {
|
|
121
|
+
"@language": "en",
|
|
122
|
+
"@vocab": "https://schema.org/",
|
|
123
|
+
"sc": "https://schema.org/",
|
|
124
|
+
"cr": "http://mlcommons.org/croissant/",
|
|
125
|
+
"rai": "http://mlcommons.org/croissant/RAI/",
|
|
126
|
+
"dct": "http://purl.org/dc/terms/",
|
|
127
|
+
"conformsTo": "dct:conformsTo"
|
|
128
|
+
},
|
|
129
|
+
"@type": "sc:Dataset",
|
|
130
|
+
"conformsTo": "http://mlcommons.org/croissant/1.1",
|
|
131
|
+
"@id": "https://huggingface.co/datasets/openai/gsm8k",
|
|
132
|
+
"name": "GSM8K",
|
|
133
|
+
"url": "https://github.com/openai/grade-school-math",
|
|
134
|
+
"publisher": {"@type": "Organization", "name": "OpenAI"},
|
|
135
|
+
"datePublished": "2021-11-18",
|
|
136
|
+
"rai:dataCollection": "Problems were initially collected by hiring freelance ...",
|
|
137
|
+
"rai:dataCollectionType": "Manual Human Curator, Others",
|
|
138
|
+
"rai:dataAnnotationPlatform": "Upwork (upwork.com) for initial collection; Surge AI ...",
|
|
139
|
+
"rai:dataBiases": "Seed questions used to assist contractors were automatically ...",
|
|
140
|
+
// and description, inLanguage, cr:citeAs, creator, cr:isLiveDataset and 9 more rai: fields
|
|
141
|
+
}
|
|
142
|
+
```
|
|
143
|
+
|
|
144
|
+
</details>
|
|
145
|
+
|
|
146
|
+
| Option | What it does |
|
|
147
|
+
|---|---|
|
|
148
|
+
| `-o my_dataset.json` | choose the output file |
|
|
149
|
+
| `--method react` | use another system (list them with `croissantminer methods`) |
|
|
150
|
+
| `--hf-id org/name` | give the dataset's Hugging Face id, so the agentic systems can check its license and URL |
|
|
151
|
+
| `--card README.md` | read the dataset card together with the paper |
|
|
152
|
+
| `--fields values.json` | also save the extracted values with their supporting quotes |
|
|
153
|
+
| `--merge-into org/name` | add the fields to the dataset's Croissant file on Hugging Face (see [Using the file](https://github.com/berkearda/croissantminer#using-the-file)) |
|
|
154
|
+
|
|
155
|
+
`croissantminer validate my_dataset.json` checks any Croissant file with the MLCommons validator.
|
|
156
|
+
|
|
157
|
+
From Python:
|
|
158
|
+
|
|
159
|
+
```python
|
|
160
|
+
from croissantminer import extract
|
|
161
|
+
|
|
162
|
+
result = extract("paper.pdf") # method="single-pass" by default
|
|
163
|
+
print(result.summary()) # Found 22 of 30 fields (core 9 of 10, Responsible AI 13 of 20)
|
|
164
|
+
result.fields["rai:dataCollection"] # one extracted value
|
|
165
|
+
result.croissant # the Croissant 1.1 file as a dict
|
|
166
|
+
```
|
|
167
|
+
|
|
168
|
+
To change the code, install from a clone instead: `git clone https://github.com/berkearda/croissantminer`, then
|
|
169
|
+
`pip install -e ".[validate]"` in that folder.
|
|
170
|
+
|
|
171
|
+
## Which method to choose
|
|
172
|
+
|
|
173
|
+
<p align="center"><img src="https://raw.githubusercontent.com/berkearda/croissantminer/main/docs/figures/architectures.png" width="860" alt="The five system designs: single-pass extraction, Parallel Specialists, Triage + Critique, Locator-Extractor and a ReAct agent"></p>
|
|
174
|
+
<p align="center"><sub>The five system designs, from the paper. Single-pass reads the whole paper in one model call; the four
|
|
175
|
+
agentic systems split the work into steps.</sub></p>
|
|
176
|
+
|
|
177
|
+
All six methods are systems from the paper, with the same code and settings. Score: the composite over the 30
|
|
178
|
+
fields on the 88 test papers (see [Results](https://github.com/berkearda/croissantminer#results)). Time and cost: one run on the 22-page GSM8K paper.
|
|
179
|
+
|
|
180
|
+
| Method | Model | Score | Time | Cost | API key |
|
|
181
|
+
|---|---|---|---|---|---|
|
|
182
|
+
| `single-pass` (default) | Claude Sonnet 4.6 | **0.709** | 35 s | $0.06 | `ANTHROPIC_API_KEY` |
|
|
183
|
+
| `single-pass-gpt` | GPT-5.4 | 0.665 | about 30 s | not measured | `OPENAI_API_KEY` |
|
|
184
|
+
| `react` | Claude Sonnet 4.6 | 0.652 | 80 s | $0.17 | `ANTHROPIC_API_KEY` |
|
|
185
|
+
| `parallel-specialists` | Claude Sonnet 4.6 | 0.647 | 15 s | $0.42 | `ANTHROPIC_API_KEY` |
|
|
186
|
+
| `triage-critique` | Claude Sonnet 4.6 | 0.624 | 35 s | $0.08 | `ANTHROPIC_API_KEY` |
|
|
187
|
+
| `locator-extractor` | Claude Sonnet 4.6 | 0.566 | 40 s | $0.09 | `ANTHROPIC_API_KEY` |
|
|
188
|
+
|
|
189
|
+
Start with `single-pass`: it is the most accurate and among the cheapest. `triage-critique` and
|
|
190
|
+
`locator-extractor` return a supporting quote for most values (`--fields`), which makes checking faster, and
|
|
191
|
+
`react` gives a reason for each field it leaves empty.
|
|
192
|
+
|
|
193
|
+
## Using the file
|
|
194
|
+
|
|
195
|
+
The file describes the dataset: the core fields and the Responsible AI fields the paper supports. It does not list
|
|
196
|
+
the data files and their columns (Croissant's `distribution` and `recordSet`), which a data host generates from the
|
|
197
|
+
files themselves. Hosts such as Hugging Face, Kaggle and OpenML publish such a file for their datasets, but without
|
|
198
|
+
the Responsible AI fields. CroissantMiner adds its fields to the host's file:
|
|
199
|
+
|
|
200
|
+
```bash
|
|
201
|
+
croissantminer extract paper.pdf --merge-into org/name # extract, then merge into the file Hugging Face generates
|
|
202
|
+
croissantminer merge org/name paper.croissant.json # merge a file you already extracted
|
|
203
|
+
croissantminer merge host_croissant.json paper.croissant.json # any host's file, by path or URL
|
|
204
|
+
```
|
|
205
|
+
|
|
206
|
+
The merged file keeps everything the host wrote (name, URL, license, files and columns) and adds the Responsible AI
|
|
207
|
+
fields and any core field the host lacks; a value the host already has is never replaced. On GSM8K, the merged file
|
|
208
|
+
kept Hugging Face's 3 data files and 4 record sets, gained 12 Responsible AI fields and passed the validator. A
|
|
209
|
+
private or gated Hugging Face dataset needs `HF_TOKEN`.
|
|
210
|
+
|
|
211
|
+
**Submitting a dataset to NeurIPS?** The [step-by-step guide](https://github.com/berkearda/croissantminer/blob/main/docs/neurips.md) covers the Croissant file the
|
|
212
|
+
Evaluations and Datasets Track requires, including the three Responsible AI items you add yourself.
|
|
213
|
+
|
|
214
|
+
## Before you publish the file
|
|
215
|
+
|
|
216
|
+
- **Check every value against the paper.** The fields are drafts. Typical mistakes are a value the paper does
|
|
217
|
+
not state, a detail from a related dataset, or, for an anonymous submission, the page header taken as the
|
|
218
|
+
publisher.
|
|
219
|
+
- **Empty fields are left out** of the Croissant file, never filled with placeholders. Add what you know.
|
|
220
|
+
- **The validator checks the format, not the content.** A file that passes can still contain wrong values.
|
|
221
|
+
- **Your paper is sent to the model provider** (Anthropic or OpenAI) under your API key and their terms.
|
|
222
|
+
- **API keys** are read from the environment or a `.env` file, never from the command line, so they do not end
|
|
223
|
+
up in your shell history.
|
|
224
|
+
|
|
225
|
+
## The benchmark
|
|
226
|
+
|
|
227
|
+
<p align="center"><img src="https://raw.githubusercontent.com/berkearda/croissantminer/main/docs/figures/pipeline.png" width="900" alt="How the benchmark was built: corpus, extraction, human annotation, adjudication to gold"></p>
|
|
228
|
+
<p align="center"><sub>How the benchmark was built, from the paper: 602 dataset papers, drafts of all 30 fields by Claude Sonnet 4.5,
|
|
229
|
+
9,595 ratings by 22 annotators, and a majority vote or an expert decision for each of the 3,060 gold cells.</sub></p>
|
|
230
|
+
|
|
231
|
+
- **Papers:** 602 dataset papers. 102 have human-validated gold annotations (3,060 cells, 22 annotators) and 500
|
|
232
|
+
have LLM-generated silver annotations. The 102 gold papers are split into 14 development and 88 test papers.
|
|
233
|
+
- **Systems:** single-pass extraction and four agentic architectures (ReAct, Parallel Specialists,
|
|
234
|
+
Triage + Critique, Locator-Extractor), each with several LLM backbones.
|
|
235
|
+
- **Evaluation:** rule-based scoring for the 10 core fields, an LLM judge (GLM-5) for the 20 RAI fields, and tests
|
|
236
|
+
that check the scorer against the numbers in the paper.
|
|
237
|
+
- **Data:** the annotations, system outputs and judge verdicts are on
|
|
238
|
+
[Hugging Face](https://huggingface.co/datasets/bearda/croissantminer) and in `data/` (see `data/README.md`).
|
|
239
|
+
|
|
240
|
+
### Results
|
|
241
|
+
|
|
242
|
+
Test split (88 papers). *Core* averages the 10 core fields, *RAI* the 20 RAI fields, and *Composite* weights all
|
|
243
|
+
30 fields equally; 95% confidence intervals come from 2,000 bootstrap samples over papers. The gold annotations
|
|
244
|
+
were first drafted by Claude Sonnet 4.5 and then checked and corrected by annotators, so Anthropic-family
|
|
245
|
+
systems are marked with \*. Claude Sonnet 4.5 itself is shown for reference and not ranked.
|
|
246
|
+
|
|
247
|
+
| System | Architecture | Core | RAI | Composite [95% CI] |
|
|
248
|
+
|---|---|---|---|---|
|
|
249
|
+
| Claude Sonnet 4.6\* | Single-pass | 0.752 | 0.687 | **0.709** [0.688, 0.729] |
|
|
250
|
+
| Claude Opus 4.7\* | Single-pass | 0.676 | 0.711 | 0.699 [0.665, 0.732] |
|
|
251
|
+
| GPT-5.4 | Single-pass | 0.653 | 0.671 | 0.665 [0.648, 0.692] |
|
|
252
|
+
| Qwen 3.6 35B-A3B | Single-pass | 0.698 | 0.601 | 0.634 [0.615, 0.654] |
|
|
253
|
+
| GLM-5.1 | Single-pass | 0.675 | 0.599 | 0.625 [0.604, 0.645] |
|
|
254
|
+
| Gemini 2.5 Flash | Single-pass | 0.615 | 0.616 | 0.616 [0.592, 0.639] |
|
|
255
|
+
| GPT-5.4 Mini | Single-pass | 0.561 | 0.614 | 0.596 [0.573, 0.620] |
|
|
256
|
+
| Gemini 3.1 Pro Preview | Single-pass | 0.577 | 0.596 | 0.590 [0.571, 0.609] |
|
|
257
|
+
| DeepSeek V3.2 | Single-pass | 0.617 | 0.555 | 0.575 [0.547, 0.604] |
|
|
258
|
+
| Mistral Small 4 | Single-pass | 0.616 | 0.482 | 0.527 [0.510, 0.544] |
|
|
259
|
+
| Llama 4 Scout 17B | Single-pass | 0.506 | 0.334 | 0.391 [0.374, 0.409] |
|
|
260
|
+
| ReAct (Sonnet 4.6)\* | ReAct | 0.734 | 0.610 | 0.652 [0.626, 0.687] |
|
|
261
|
+
| ReAct (GPT-5.4) | ReAct | 0.688 | 0.603 | 0.631 [0.601, 0.663] |
|
|
262
|
+
| ReAct (Gemini 3.1 Pro) | ReAct | 0.723 | 0.511 | 0.582 [0.556, 0.605] |
|
|
263
|
+
| Parallel Specialists (Sonnet 4.6)\* | Parallel Specialists | 0.699 | 0.621 | 0.647 [0.626, 0.667] |
|
|
264
|
+
| Parallel Specialists (GPT-5.4) | Parallel Specialists | 0.627 | 0.570 | 0.589 [0.572, 0.611] |
|
|
265
|
+
| Parallel Specialists (Gemini 3.1 Pro) | Parallel Specialists | 0.607 | 0.505 | 0.539 [0.518, 0.559] |
|
|
266
|
+
| Triage + Critique (Sonnet 4.6)\* | Triage + Critique | 0.675 | 0.599 | 0.624 [0.606, 0.653] |
|
|
267
|
+
| Triage + Critique (GPT-5.4) | Triage + Critique | 0.592 | 0.540 | 0.557 [0.537, 0.585] |
|
|
268
|
+
| Triage + Critique (Gemini 3.1 Pro) | Triage + Critique | 0.513 | 0.397 | 0.436 [0.415, 0.457] |
|
|
269
|
+
| Locator-Extractor (Sonnet 4.6)\* | Locator-Extractor | 0.643 | 0.528 | 0.566 [0.543, 0.592] |
|
|
270
|
+
| Locator-Extractor (GPT-5.4) | Locator-Extractor | 0.523 | 0.492 | 0.502 [0.481, 0.528] |
|
|
271
|
+
| Locator-Extractor (Gemini 3.1 Pro + GPT-5.4 Mini) | Locator-Extractor | 0.560 | 0.436 | 0.478 [0.456, 0.502] |
|
|
272
|
+
| Locator-Extractor (Gemini 3.1 Pro) | Locator-Extractor | 0.500 | 0.422 | 0.448 [0.423, 0.471] |
|
|
273
|
+
| *Claude Sonnet 4.5\* (reference)* | *Single-pass* | *0.903* | *0.840* | *0.861 [0.830, 0.893]* |
|
|
274
|
+
|
|
275
|
+
## Reproducing the paper
|
|
276
|
+
|
|
277
|
+
### Installation
|
|
278
|
+
|
|
279
|
+
The paper's environment uses Python 3.10 or 3.11 and the pinned versions in `requirements.txt`:
|
|
280
|
+
|
|
281
|
+
```bash
|
|
282
|
+
git clone https://github.com/berkearda/croissantminer
|
|
283
|
+
cd croissantminer
|
|
284
|
+
pip install -r requirements.txt
|
|
285
|
+
pip install -e .
|
|
286
|
+
```
|
|
287
|
+
|
|
288
|
+
API keys are only needed to run the systems or the judge. Copy `.env.example` to `.env` and fill in the keys for
|
|
289
|
+
the providers you use.
|
|
290
|
+
|
|
291
|
+
### Checking the numbers
|
|
292
|
+
|
|
293
|
+
No API keys and no cost: the scores are recomputed from the stored system outputs
|
|
294
|
+
(`data/extractions/`), judge verdicts (`data/judged/`) and gold annotations
|
|
295
|
+
(`data/annotations/gold.parquet`), which are included in this repository.
|
|
296
|
+
|
|
297
|
+
1. Check Table 2 and the per-field tables in the appendix against the published numbers:
|
|
298
|
+
```bash
|
|
299
|
+
make reproduce
|
|
300
|
+
```
|
|
301
|
+
2. Print Table 2 (Core, RAI and Composite with 95% confidence intervals, grouped as in the paper):
|
|
302
|
+
```bash
|
|
303
|
+
make table2
|
|
304
|
+
```
|
|
305
|
+
3. Run the pairwise significance tests (paired bootstrap, Wilcoxon and McNemar with BH-FDR correction):
|
|
306
|
+
```bash
|
|
307
|
+
make significance
|
|
308
|
+
```
|
|
309
|
+
|
|
310
|
+
### Running the systems and the scoring rules
|
|
311
|
+
|
|
312
|
+
[docs/reproducing.md](https://github.com/berkearda/croissantminer/blob/main/docs/reproducing.md) explains how to re-run each system on the benchmark (this needs API keys
|
|
313
|
+
and the benchmark PDFs) and gives the exact scoring rules: rule-based scores for the 10 core fields, the GLM-5 judge
|
|
314
|
+
for the 20 RAI fields, and how empty values are scored.
|
|
315
|
+
|
|
316
|
+
### Tests
|
|
317
|
+
|
|
318
|
+
```bash
|
|
319
|
+
make test
|
|
320
|
+
```
|
|
321
|
+
|
|
322
|
+
About 120 tests, about 10 seconds, no API keys. They cover the scoring rules, the handling of model output, the
|
|
323
|
+
command line and the Croissant output, and the check that Table 2 and Tables 5 and 6 are reproduced exactly.
|
|
324
|
+
GitHub Actions runs them on every push, and the tool's tests on Python 3.10 to 3.13.
|
|
325
|
+
|
|
326
|
+
## Extending CroissantMiner
|
|
327
|
+
|
|
328
|
+
- **A new method or model for the tool:** methods are registered in `croissantminer/methods.py` (`METHODS` and
|
|
329
|
+
`run`), model backbones in `croissantminer/systems/helpers.py` (`MODELS`), and the Croissant file is built in
|
|
330
|
+
`croissantminer/croissant.py`.
|
|
331
|
+
- **A new system on the benchmark:** `make evaluate OUTPUTS=folder NAME=name` scores it with the paper's scorer and
|
|
332
|
+
judge model; the [leaderboard](https://github.com/berkearda/croissantminer/blob/main/leaderboard/README.md) explains the format and how to add your entry.
|
|
333
|
+
- **A wrong extraction, a bug or an idea:** open an [issue](https://github.com/berkearda/croissantminer/issues/new/choose)
|
|
334
|
+
or a [discussion](https://github.com/berkearda/croissantminer/discussions).
|
|
335
|
+
|
|
336
|
+
## Repository layout
|
|
337
|
+
|
|
338
|
+
| Path | Contents |
|
|
339
|
+
|---|---|
|
|
340
|
+
| `croissantminer/` | Package: command line and Python API (`cli.py`, `api.py`), the six methods (`methods.py`) and the systems' code (`systems/`), the Croissant file (`croissant.py`), the extraction prompt, PDF reading and the ReAct agent |
|
|
341
|
+
| `scripts/` | Benchmark runs of the systems, the judge, tables and figures (guide in `scripts/README.md`) |
|
|
342
|
+
| `evaluation/` | Field metrics and system registry used by the scorer |
|
|
343
|
+
| `data/` | Gold annotations, judge verdicts and system outputs |
|
|
344
|
+
| `silver/` | Selection and extraction of the 500 silver papers |
|
|
345
|
+
| `tests/` | Tests and the published numbers they check against |
|
|
346
|
+
| `hf_space/` | The Hugging Face Space demo |
|
|
347
|
+
| `docs/` | Reproduction guide (`reproducing.md`), README figures and the ReAct agent's prompts |
|
|
348
|
+
| `legacy/` | Early prototype and experiments from before the paper, kept for reference and not maintained |
|
|
349
|
+
|
|
350
|
+
The scripts that read the named annotation sheets are not included, to protect the annotators'
|
|
351
|
+
privacy; `data/annotations/gold.parquet` and `data/annotations/iaa.parquet` are their output.
|
|
352
|
+
Comments that cite `decisions.md` or task numbers (`T-###`) refer to our
|
|
353
|
+
internal project log, which is not included.
|
|
354
|
+
|
|
355
|
+
## Community
|
|
356
|
+
|
|
357
|
+
Questions and ideas go to [Discussions](https://github.com/berkearda/croissantminer/discussions), bugs and wrong
|
|
358
|
+
extractions to [issues](https://github.com/berkearda/croissantminer/issues). Please read
|
|
359
|
+
[CONTRIBUTING.md](https://github.com/berkearda/croissantminer/blob/main/CONTRIBUTING.md) before opening a pull request. Everyone taking part follows the
|
|
360
|
+
[code of conduct](https://github.com/berkearda/croissantminer/blob/main/CODE_OF_CONDUCT.md); security problems are reported as described in [SECURITY.md](https://github.com/berkearda/croissantminer/blob/main/SECURITY.md).
|
|
361
|
+
Changes are listed in [CHANGELOG.md](https://github.com/berkearda/croissantminer/blob/main/CHANGELOG.md).
|
|
362
|
+
|
|
363
|
+
## Citation
|
|
364
|
+
|
|
365
|
+
```bibtex
|
|
366
|
+
@inproceedings{arda2026croissantminer,
|
|
367
|
+
title = {CroissantMiner: Automated Extraction and Validation of Croissant Metadata for ML Datasets},
|
|
368
|
+
author = {Arda, Berke and Yavuz, Ahmetcan and Gerry, Paul and Lobentanzer, Sebastian and
|
|
369
|
+
Sarwar, Nobin and Giner-Miguelez, Joan and Chen, Kongtao and Zhang, Luyao and
|
|
370
|
+
Sachan, Mrinmaya and Akhtar, Mubashara},
|
|
371
|
+
booktitle = {Advances in Neural Information Processing Systems (Evaluations and Datasets Track)},
|
|
372
|
+
year = {2026}
|
|
373
|
+
}
|
|
374
|
+
```
|
|
375
|
+
|
|
376
|
+
## License
|
|
377
|
+
|
|
378
|
+
Code: MIT (see `LICENSE`). Annotations: CC BY 4.0 (see the dataset card). The papers remain under their
|
|
379
|
+
authors' licenses.
|