croissantminer 0.2.0__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (61) hide show
  1. croissantminer-0.2.0/LICENSE +21 -0
  2. croissantminer-0.2.0/PKG-INFO +379 -0
  3. croissantminer-0.2.0/README.md +330 -0
  4. croissantminer-0.2.0/croissantminer/__init__.py +41 -0
  5. croissantminer-0.2.0/croissantminer/__main__.py +5 -0
  6. croissantminer-0.2.0/croissantminer/api.py +106 -0
  7. croissantminer-0.2.0/croissantminer/cli.py +198 -0
  8. croissantminer-0.2.0/croissantminer/confidence.py +201 -0
  9. croissantminer-0.2.0/croissantminer/config.py +153 -0
  10. croissantminer-0.2.0/croissantminer/croissant.py +240 -0
  11. croissantminer-0.2.0/croissantminer/evaluator.py +449 -0
  12. croissantminer-0.2.0/croissantminer/extractor.py +293 -0
  13. croissantminer-0.2.0/croissantminer/field_filter.py +120 -0
  14. croissantminer-0.2.0/croissantminer/field_types.py +278 -0
  15. croissantminer-0.2.0/croissantminer/groundtruth_parser.py +72 -0
  16. croissantminer-0.2.0/croissantminer/groundtruth_parser_md.py +222 -0
  17. croissantminer-0.2.0/croissantminer/llm_evaluator.py +546 -0
  18. croissantminer-0.2.0/croissantminer/markdown_generator.py +288 -0
  19. croissantminer-0.2.0/croissantminer/methods.py +324 -0
  20. croissantminer-0.2.0/croissantminer/metrics.py +911 -0
  21. croissantminer-0.2.0/croissantminer/models/__init__.py +5 -0
  22. croissantminer-0.2.0/croissantminer/models/base.py +18 -0
  23. croissantminer-0.2.0/croissantminer/models/claude_model.py +321 -0
  24. croissantminer-0.2.0/croissantminer/models/factory.py +84 -0
  25. croissantminer-0.2.0/croissantminer/models/gemini_model.py +145 -0
  26. croissantminer-0.2.0/croissantminer/models/openai_model.py +62 -0
  27. croissantminer-0.2.0/croissantminer/models/qwen_model.py +279 -0
  28. croissantminer-0.2.0/croissantminer/pdf/__init__.py +22 -0
  29. croissantminer-0.2.0/croissantminer/pdf/processor.py +1115 -0
  30. croissantminer-0.2.0/croissantminer/pdf/reader.py +71 -0
  31. croissantminer-0.2.0/croissantminer/react_agent/__init__.py +7 -0
  32. croissantminer-0.2.0/croissantminer/react_agent/agent.py +761 -0
  33. croissantminer-0.2.0/croissantminer/react_agent/audit.py +138 -0
  34. croissantminer-0.2.0/croissantminer/react_agent/runner.py +231 -0
  35. croissantminer-0.2.0/croissantminer/react_agent/schemas.py +72 -0
  36. croissantminer-0.2.0/croissantminer/react_agent/tools.py +395 -0
  37. croissantminer-0.2.0/croissantminer/relevance.py +108 -0
  38. croissantminer-0.2.0/croissantminer/reporter.py +490 -0
  39. croissantminer-0.2.0/croissantminer/systems/__init__.py +12 -0
  40. croissantminer-0.2.0/croissantminer/systems/helpers.py +452 -0
  41. croissantminer-0.2.0/croissantminer/systems/locator_extractor.py +1233 -0
  42. croissantminer-0.2.0/croissantminer/systems/sections.py +588 -0
  43. croissantminer-0.2.0/croissantminer/systems/specialist_prompts.py +326 -0
  44. croissantminer-0.2.0/croissantminer/systems/specialists.py +305 -0
  45. croissantminer-0.2.0/croissantminer/systems/triage_critique.py +967 -0
  46. croissantminer-0.2.0/croissantminer/systems/validate.py +90 -0
  47. croissantminer-0.2.0/croissantminer/unifier.py +227 -0
  48. croissantminer-0.2.0/croissantminer.egg-info/PKG-INFO +379 -0
  49. croissantminer-0.2.0/croissantminer.egg-info/SOURCES.txt +59 -0
  50. croissantminer-0.2.0/croissantminer.egg-info/dependency_links.txt +1 -0
  51. croissantminer-0.2.0/croissantminer.egg-info/entry_points.txt +2 -0
  52. croissantminer-0.2.0/croissantminer.egg-info/requires.txt +27 -0
  53. croissantminer-0.2.0/croissantminer.egg-info/top_level.txt +1 -0
  54. croissantminer-0.2.0/pyproject.toml +77 -0
  55. croissantminer-0.2.0/setup.cfg +4 -0
  56. croissantminer-0.2.0/tests/test_cli.py +105 -0
  57. croissantminer-0.2.0/tests/test_croissant_output.py +108 -0
  58. croissantminer-0.2.0/tests/test_extraction_output.py +83 -0
  59. croissantminer-0.2.0/tests/test_leaderboard.py +54 -0
  60. croissantminer-0.2.0/tests/test_scoring_rules.py +84 -0
  61. croissantminer-0.2.0/tests/test_table2_reproduction.py +74 -0
@@ -0,0 +1,21 @@
1
+ MIT License
2
+
3
+ Copyright (c) 2026 The CroissantMiner Authors
4
+
5
+ Permission is hereby granted, free of charge, to any person obtaining a copy
6
+ of this software and associated documentation files (the "Software"), to deal
7
+ in the Software without restriction, including without limitation the rights
8
+ to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
9
+ copies of the Software, and to permit persons to whom the Software is
10
+ furnished to do so, subject to the following conditions:
11
+
12
+ The above copyright notice and this permission notice shall be included in all
13
+ copies or substantial portions of the Software.
14
+
15
+ THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
16
+ IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
17
+ FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
18
+ AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
19
+ LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
20
+ OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
21
+ SOFTWARE.
@@ -0,0 +1,379 @@
1
+ Metadata-Version: 2.4
2
+ Name: croissantminer
3
+ Version: 0.2.0
4
+ Summary: Extract Croissant metadata, including the Responsible AI fields, from ML dataset papers
5
+ Author-email: Berke Arda <bearda@ethz.ch>
6
+ License: MIT
7
+ Project-URL: Homepage, https://github.com/berkearda/croissantminer
8
+ Project-URL: Demo, https://huggingface.co/spaces/bearda/croissantminer
9
+ Project-URL: Dataset, https://huggingface.co/datasets/bearda/croissantminer
10
+ Project-URL: Issues, https://github.com/berkearda/croissantminer/issues
11
+ Project-URL: Changelog, https://github.com/berkearda/croissantminer/blob/main/CHANGELOG.md
12
+ Keywords: machine-learning,metadata,croissant,responsible-ai,datasets,llm,documentation
13
+ Classifier: Development Status :: 4 - Beta
14
+ Classifier: Intended Audience :: Science/Research
15
+ Classifier: License :: OSI Approved :: MIT License
16
+ Classifier: Programming Language :: Python :: 3
17
+ Classifier: Programming Language :: Python :: 3.10
18
+ Classifier: Programming Language :: Python :: 3.11
19
+ Classifier: Programming Language :: Python :: 3.12
20
+ Classifier: Programming Language :: Python :: 3.13
21
+ Classifier: Topic :: Scientific/Engineering :: Artificial Intelligence
22
+ Requires-Python: >=3.10
23
+ Description-Content-Type: text/markdown
24
+ License-File: LICENSE
25
+ Requires-Dist: anthropic<1,>=0.72.0
26
+ Requires-Dist: openai<2,>=1.76.0
27
+ Requires-Dist: PyPDF2>=3.0.1
28
+ Requires-Dist: requests
29
+ Requires-Dist: tqdm
30
+ Requires-Dist: python-dotenv
31
+ Provides-Extra: validate
32
+ Requires-Dist: mlcroissant<2,>=1.0; extra == "validate"
33
+ Provides-Extra: eval
34
+ Requires-Dist: pandas>=2.0.3; extra == "eval"
35
+ Requires-Dist: pyarrow>=14.0; extra == "eval"
36
+ Requires-Dist: numpy>=1.24.4; extra == "eval"
37
+ Requires-Dist: scipy>=1.10.1; extra == "eval"
38
+ Requires-Dist: scikit-learn>=1.3; extra == "eval"
39
+ Requires-Dist: matplotlib>=3.7; extra == "eval"
40
+ Requires-Dist: openpyxl>=3.1.0; extra == "eval"
41
+ Requires-Dist: regex; extra == "eval"
42
+ Requires-Dist: PyMuPDF>=1.24.0; extra == "eval"
43
+ Provides-Extra: demo
44
+ Requires-Dist: gradio<7,>=6.14; extra == "demo"
45
+ Provides-Extra: dev
46
+ Requires-Dist: pytest>=8.0; extra == "dev"
47
+ Requires-Dist: ruff>=0.5.0; extra == "dev"
48
+ Dynamic: license-file
49
+
50
+ <p align="center"><img src="https://raw.githubusercontent.com/berkearda/croissantminer/main/assets/croissantminer-logo.svg" width="520" alt="CroissantMiner"></p>
51
+
52
+ <p align="center">
53
+ Extract <a href="https://github.com/mlcommons/croissant">Croissant</a> metadata, including the 20 Responsible AI fields,
54
+ from the paper that introduces an ML dataset.
55
+ </p>
56
+
57
+ <p align="center">
58
+ <a href="https://huggingface.co/spaces/bearda/croissantminer"><img alt="Demo" src="https://img.shields.io/badge/demo-Hugging%20Face%20Space-ffcc4d"></a>
59
+ <a href="https://huggingface.co/datasets/bearda/croissantminer"><img alt="Dataset" src="https://img.shields.io/badge/dataset-Hugging%20Face-ffcc4d"></a>
60
+ <a href="https://github.com/berkearda/croissantminer/actions/workflows/tests.yml"><img alt="Tests" src="https://github.com/berkearda/croissantminer/actions/workflows/tests.yml/badge.svg"></a>
61
+ <img alt="Python 3.10 to 3.13" src="https://img.shields.io/badge/python-3.10%20to%203.13-3776ab">
62
+ <a href="https://github.com/berkearda/croissantminer/blob/main/LICENSE"><img alt="License: MIT" src="https://img.shields.io/badge/license-MIT-2ea44f"></a>
63
+ </p>
64
+
65
+ Code, benchmark and systems of **CroissantMiner: Automated Extraction and Validation of Croissant Metadata for ML
66
+ Datasets** (NeurIPS 2026, Evaluations and Datasets Track). Paper: arXiv link follows.
67
+
68
+ **News**
69
+ - **1 Oct 2026:** on PyPI (`pip install croissantminer`), with `--merge-into` to add the fields to a dataset's
70
+ Croissant file on Hugging Face, a
71
+ [guide for NeurIPS dataset submissions](https://github.com/berkearda/croissantminer/blob/main/docs/neurips.md) and a
72
+ [leaderboard](https://github.com/berkearda/croissantminer/blob/main/leaderboard/README.md) open to new systems.
73
+ - **30 Sep 2026:** `croissantminer extract`, one command from a paper to a Croissant file.
74
+ - **28 Sep 2026:** code, [dataset](https://huggingface.co/datasets/bearda/croissantminer) and
75
+ [demo](https://huggingface.co/spaces/bearda/croissantminer) released.
76
+ - **24 Sep 2026:** accepted at NeurIPS 2026 (Evaluations and Datasets Track).
77
+
78
+ Documenting a dataset in the [Croissant](https://github.com/mlcommons/croissant) format, including its Responsible
79
+ AI (RAI) fields, takes time, and venues such as the NeurIPS Evaluations and Datasets Track ask for it.
80
+ CroissantMiner reads the paper that introduces a dataset and drafts all 30 fields of the Croissant 1.1 schema: 10
81
+ core fields (name, license, creators and others) and 20 RAI fields (how the data was collected and annotated,
82
+ known biases, limitations, intended uses and others). You check the draft and publish it.
83
+
84
+ - **For dataset authors:** one command turns a paper into a Croissant file that passes the MLCommons validator.
85
+ - **For researchers:** a benchmark of 602 dataset papers with human gold annotations for 102 of them, the outputs
86
+ and scores of 24 extraction systems, and a [leaderboard](https://github.com/berkearda/croissantminer/blob/main/leaderboard/README.md) that scores new ones with the
87
+ paper's scorer.
88
+
89
+ ## Try it in your browser
90
+
91
+ The [demo](https://huggingface.co/spaces/bearda/croissantminer) runs the six systems below on a PDF you upload.
92
+ It needs your own Anthropic or OpenAI API key, which is sent only to that provider and not stored.
93
+
94
+ <p align="center"><img src="https://raw.githubusercontent.com/berkearda/croissantminer/main/docs/figures/demo.png" width="760" alt="The demo after extracting the GSM8K paper with Triage + Critique: 19 of 30 fields, each with a supporting quote from the paper"></p>
95
+
96
+ ## Quick start
97
+
98
+ Python 3.10 to 3.13.
99
+
100
+ ```bash
101
+ pip install "croissantminer[validate]" # the extraction tool and the Croissant validator
102
+ export ANTHROPIC_API_KEY=... # or put it in a .env file in the folder you run from
103
+ croissantminer extract paper.pdf
104
+ ```
105
+
106
+ ```text
107
+ Extracting with Single-pass · Claude Sonnet 4.6 (usually about 30 s)...
108
+ Found 22 of 30 fields (core 9 of 10, Responsible AI 13 of 20) with single-pass in 35 s, about $0.06.
109
+ Not found: license, rai:dataCollectionMissingData, rai:dataCollectionTimeframe, ...
110
+ Wrote: paper.croissant.json (Croissant 1.1)
111
+ Check: passes the mlcroissant validator (2 recommended properties missing)
112
+ These are drafts by a language model: check each value against the paper before publishing.
113
+ ```
114
+
115
+ <details>
116
+ <summary>Example output for GSM8K (<code>--hf-id openai/gsm8k</code>), shortened</summary>
117
+
118
+ ```jsonc
119
+ {
120
+ "@context": {
121
+ "@language": "en",
122
+ "@vocab": "https://schema.org/",
123
+ "sc": "https://schema.org/",
124
+ "cr": "http://mlcommons.org/croissant/",
125
+ "rai": "http://mlcommons.org/croissant/RAI/",
126
+ "dct": "http://purl.org/dc/terms/",
127
+ "conformsTo": "dct:conformsTo"
128
+ },
129
+ "@type": "sc:Dataset",
130
+ "conformsTo": "http://mlcommons.org/croissant/1.1",
131
+ "@id": "https://huggingface.co/datasets/openai/gsm8k",
132
+ "name": "GSM8K",
133
+ "url": "https://github.com/openai/grade-school-math",
134
+ "publisher": {"@type": "Organization", "name": "OpenAI"},
135
+ "datePublished": "2021-11-18",
136
+ "rai:dataCollection": "Problems were initially collected by hiring freelance ...",
137
+ "rai:dataCollectionType": "Manual Human Curator, Others",
138
+ "rai:dataAnnotationPlatform": "Upwork (upwork.com) for initial collection; Surge AI ...",
139
+ "rai:dataBiases": "Seed questions used to assist contractors were automatically ...",
140
+ // and description, inLanguage, cr:citeAs, creator, cr:isLiveDataset and 9 more rai: fields
141
+ }
142
+ ```
143
+
144
+ </details>
145
+
146
+ | Option | What it does |
147
+ |---|---|
148
+ | `-o my_dataset.json` | choose the output file |
149
+ | `--method react` | use another system (list them with `croissantminer methods`) |
150
+ | `--hf-id org/name` | give the dataset's Hugging Face id, so the agentic systems can check its license and URL |
151
+ | `--card README.md` | read the dataset card together with the paper |
152
+ | `--fields values.json` | also save the extracted values with their supporting quotes |
153
+ | `--merge-into org/name` | add the fields to the dataset's Croissant file on Hugging Face (see [Using the file](https://github.com/berkearda/croissantminer#using-the-file)) |
154
+
155
+ `croissantminer validate my_dataset.json` checks any Croissant file with the MLCommons validator.
156
+
157
+ From Python:
158
+
159
+ ```python
160
+ from croissantminer import extract
161
+
162
+ result = extract("paper.pdf") # method="single-pass" by default
163
+ print(result.summary()) # Found 22 of 30 fields (core 9 of 10, Responsible AI 13 of 20)
164
+ result.fields["rai:dataCollection"] # one extracted value
165
+ result.croissant # the Croissant 1.1 file as a dict
166
+ ```
167
+
168
+ To change the code, install from a clone instead: `git clone https://github.com/berkearda/croissantminer`, then
169
+ `pip install -e ".[validate]"` in that folder.
170
+
171
+ ## Which method to choose
172
+
173
+ <p align="center"><img src="https://raw.githubusercontent.com/berkearda/croissantminer/main/docs/figures/architectures.png" width="860" alt="The five system designs: single-pass extraction, Parallel Specialists, Triage + Critique, Locator-Extractor and a ReAct agent"></p>
174
+ <p align="center"><sub>The five system designs, from the paper. Single-pass reads the whole paper in one model call; the four
175
+ agentic systems split the work into steps.</sub></p>
176
+
177
+ All six methods are systems from the paper, with the same code and settings. Score: the composite over the 30
178
+ fields on the 88 test papers (see [Results](https://github.com/berkearda/croissantminer#results)). Time and cost: one run on the 22-page GSM8K paper.
179
+
180
+ | Method | Model | Score | Time | Cost | API key |
181
+ |---|---|---|---|---|---|
182
+ | `single-pass` (default) | Claude Sonnet 4.6 | **0.709** | 35 s | $0.06 | `ANTHROPIC_API_KEY` |
183
+ | `single-pass-gpt` | GPT-5.4 | 0.665 | about 30 s | not measured | `OPENAI_API_KEY` |
184
+ | `react` | Claude Sonnet 4.6 | 0.652 | 80 s | $0.17 | `ANTHROPIC_API_KEY` |
185
+ | `parallel-specialists` | Claude Sonnet 4.6 | 0.647 | 15 s | $0.42 | `ANTHROPIC_API_KEY` |
186
+ | `triage-critique` | Claude Sonnet 4.6 | 0.624 | 35 s | $0.08 | `ANTHROPIC_API_KEY` |
187
+ | `locator-extractor` | Claude Sonnet 4.6 | 0.566 | 40 s | $0.09 | `ANTHROPIC_API_KEY` |
188
+
189
+ Start with `single-pass`: it is the most accurate and among the cheapest. `triage-critique` and
190
+ `locator-extractor` return a supporting quote for most values (`--fields`), which makes checking faster, and
191
+ `react` gives a reason for each field it leaves empty.
192
+
193
+ ## Using the file
194
+
195
+ The file describes the dataset: the core fields and the Responsible AI fields the paper supports. It does not list
196
+ the data files and their columns (Croissant's `distribution` and `recordSet`), which a data host generates from the
197
+ files themselves. Hosts such as Hugging Face, Kaggle and OpenML publish such a file for their datasets, but without
198
+ the Responsible AI fields. CroissantMiner adds its fields to the host's file:
199
+
200
+ ```bash
201
+ croissantminer extract paper.pdf --merge-into org/name # extract, then merge into the file Hugging Face generates
202
+ croissantminer merge org/name paper.croissant.json # merge a file you already extracted
203
+ croissantminer merge host_croissant.json paper.croissant.json # any host's file, by path or URL
204
+ ```
205
+
206
+ The merged file keeps everything the host wrote (name, URL, license, files and columns) and adds the Responsible AI
207
+ fields and any core field the host lacks; a value the host already has is never replaced. On GSM8K, the merged file
208
+ kept Hugging Face's 3 data files and 4 record sets, gained 12 Responsible AI fields and passed the validator. A
209
+ private or gated Hugging Face dataset needs `HF_TOKEN`.
210
+
211
+ **Submitting a dataset to NeurIPS?** The [step-by-step guide](https://github.com/berkearda/croissantminer/blob/main/docs/neurips.md) covers the Croissant file the
212
+ Evaluations and Datasets Track requires, including the three Responsible AI items you add yourself.
213
+
214
+ ## Before you publish the file
215
+
216
+ - **Check every value against the paper.** The fields are drafts. Typical mistakes are a value the paper does
217
+ not state, a detail from a related dataset, or, for an anonymous submission, the page header taken as the
218
+ publisher.
219
+ - **Empty fields are left out** of the Croissant file, never filled with placeholders. Add what you know.
220
+ - **The validator checks the format, not the content.** A file that passes can still contain wrong values.
221
+ - **Your paper is sent to the model provider** (Anthropic or OpenAI) under your API key and their terms.
222
+ - **API keys** are read from the environment or a `.env` file, never from the command line, so they do not end
223
+ up in your shell history.
224
+
225
+ ## The benchmark
226
+
227
+ <p align="center"><img src="https://raw.githubusercontent.com/berkearda/croissantminer/main/docs/figures/pipeline.png" width="900" alt="How the benchmark was built: corpus, extraction, human annotation, adjudication to gold"></p>
228
+ <p align="center"><sub>How the benchmark was built, from the paper: 602 dataset papers, drafts of all 30 fields by Claude Sonnet 4.5,
229
+ 9,595 ratings by 22 annotators, and a majority vote or an expert decision for each of the 3,060 gold cells.</sub></p>
230
+
231
+ - **Papers:** 602 dataset papers. 102 have human-validated gold annotations (3,060 cells, 22 annotators) and 500
232
+ have LLM-generated silver annotations. The 102 gold papers are split into 14 development and 88 test papers.
233
+ - **Systems:** single-pass extraction and four agentic architectures (ReAct, Parallel Specialists,
234
+ Triage + Critique, Locator-Extractor), each with several LLM backbones.
235
+ - **Evaluation:** rule-based scoring for the 10 core fields, an LLM judge (GLM-5) for the 20 RAI fields, and tests
236
+ that check the scorer against the numbers in the paper.
237
+ - **Data:** the annotations, system outputs and judge verdicts are on
238
+ [Hugging Face](https://huggingface.co/datasets/bearda/croissantminer) and in `data/` (see `data/README.md`).
239
+
240
+ ### Results
241
+
242
+ Test split (88 papers). *Core* averages the 10 core fields, *RAI* the 20 RAI fields, and *Composite* weights all
243
+ 30 fields equally; 95% confidence intervals come from 2,000 bootstrap samples over papers. The gold annotations
244
+ were first drafted by Claude Sonnet 4.5 and then checked and corrected by annotators, so Anthropic-family
245
+ systems are marked with \*. Claude Sonnet 4.5 itself is shown for reference and not ranked.
246
+
247
+ | System | Architecture | Core | RAI | Composite [95% CI] |
248
+ |---|---|---|---|---|
249
+ | Claude Sonnet 4.6\* | Single-pass | 0.752 | 0.687 | **0.709** [0.688, 0.729] |
250
+ | Claude Opus 4.7\* | Single-pass | 0.676 | 0.711 | 0.699 [0.665, 0.732] |
251
+ | GPT-5.4 | Single-pass | 0.653 | 0.671 | 0.665 [0.648, 0.692] |
252
+ | Qwen 3.6 35B-A3B | Single-pass | 0.698 | 0.601 | 0.634 [0.615, 0.654] |
253
+ | GLM-5.1 | Single-pass | 0.675 | 0.599 | 0.625 [0.604, 0.645] |
254
+ | Gemini 2.5 Flash | Single-pass | 0.615 | 0.616 | 0.616 [0.592, 0.639] |
255
+ | GPT-5.4 Mini | Single-pass | 0.561 | 0.614 | 0.596 [0.573, 0.620] |
256
+ | Gemini 3.1 Pro Preview | Single-pass | 0.577 | 0.596 | 0.590 [0.571, 0.609] |
257
+ | DeepSeek V3.2 | Single-pass | 0.617 | 0.555 | 0.575 [0.547, 0.604] |
258
+ | Mistral Small 4 | Single-pass | 0.616 | 0.482 | 0.527 [0.510, 0.544] |
259
+ | Llama 4 Scout 17B | Single-pass | 0.506 | 0.334 | 0.391 [0.374, 0.409] |
260
+ | ReAct (Sonnet 4.6)\* | ReAct | 0.734 | 0.610 | 0.652 [0.626, 0.687] |
261
+ | ReAct (GPT-5.4) | ReAct | 0.688 | 0.603 | 0.631 [0.601, 0.663] |
262
+ | ReAct (Gemini 3.1 Pro) | ReAct | 0.723 | 0.511 | 0.582 [0.556, 0.605] |
263
+ | Parallel Specialists (Sonnet 4.6)\* | Parallel Specialists | 0.699 | 0.621 | 0.647 [0.626, 0.667] |
264
+ | Parallel Specialists (GPT-5.4) | Parallel Specialists | 0.627 | 0.570 | 0.589 [0.572, 0.611] |
265
+ | Parallel Specialists (Gemini 3.1 Pro) | Parallel Specialists | 0.607 | 0.505 | 0.539 [0.518, 0.559] |
266
+ | Triage + Critique (Sonnet 4.6)\* | Triage + Critique | 0.675 | 0.599 | 0.624 [0.606, 0.653] |
267
+ | Triage + Critique (GPT-5.4) | Triage + Critique | 0.592 | 0.540 | 0.557 [0.537, 0.585] |
268
+ | Triage + Critique (Gemini 3.1 Pro) | Triage + Critique | 0.513 | 0.397 | 0.436 [0.415, 0.457] |
269
+ | Locator-Extractor (Sonnet 4.6)\* | Locator-Extractor | 0.643 | 0.528 | 0.566 [0.543, 0.592] |
270
+ | Locator-Extractor (GPT-5.4) | Locator-Extractor | 0.523 | 0.492 | 0.502 [0.481, 0.528] |
271
+ | Locator-Extractor (Gemini 3.1 Pro + GPT-5.4 Mini) | Locator-Extractor | 0.560 | 0.436 | 0.478 [0.456, 0.502] |
272
+ | Locator-Extractor (Gemini 3.1 Pro) | Locator-Extractor | 0.500 | 0.422 | 0.448 [0.423, 0.471] |
273
+ | *Claude Sonnet 4.5\* (reference)* | *Single-pass* | *0.903* | *0.840* | *0.861 [0.830, 0.893]* |
274
+
275
+ ## Reproducing the paper
276
+
277
+ ### Installation
278
+
279
+ The paper's environment uses Python 3.10 or 3.11 and the pinned versions in `requirements.txt`:
280
+
281
+ ```bash
282
+ git clone https://github.com/berkearda/croissantminer
283
+ cd croissantminer
284
+ pip install -r requirements.txt
285
+ pip install -e .
286
+ ```
287
+
288
+ API keys are only needed to run the systems or the judge. Copy `.env.example` to `.env` and fill in the keys for
289
+ the providers you use.
290
+
291
+ ### Checking the numbers
292
+
293
+ No API keys and no cost: the scores are recomputed from the stored system outputs
294
+ (`data/extractions/`), judge verdicts (`data/judged/`) and gold annotations
295
+ (`data/annotations/gold.parquet`), which are included in this repository.
296
+
297
+ 1. Check Table 2 and the per-field tables in the appendix against the published numbers:
298
+ ```bash
299
+ make reproduce
300
+ ```
301
+ 2. Print Table 2 (Core, RAI and Composite with 95% confidence intervals, grouped as in the paper):
302
+ ```bash
303
+ make table2
304
+ ```
305
+ 3. Run the pairwise significance tests (paired bootstrap, Wilcoxon and McNemar with BH-FDR correction):
306
+ ```bash
307
+ make significance
308
+ ```
309
+
310
+ ### Running the systems and the scoring rules
311
+
312
+ [docs/reproducing.md](https://github.com/berkearda/croissantminer/blob/main/docs/reproducing.md) explains how to re-run each system on the benchmark (this needs API keys
313
+ and the benchmark PDFs) and gives the exact scoring rules: rule-based scores for the 10 core fields, the GLM-5 judge
314
+ for the 20 RAI fields, and how empty values are scored.
315
+
316
+ ### Tests
317
+
318
+ ```bash
319
+ make test
320
+ ```
321
+
322
+ About 120 tests, about 10 seconds, no API keys. They cover the scoring rules, the handling of model output, the
323
+ command line and the Croissant output, and the check that Table 2 and Tables 5 and 6 are reproduced exactly.
324
+ GitHub Actions runs them on every push, and the tool's tests on Python 3.10 to 3.13.
325
+
326
+ ## Extending CroissantMiner
327
+
328
+ - **A new method or model for the tool:** methods are registered in `croissantminer/methods.py` (`METHODS` and
329
+ `run`), model backbones in `croissantminer/systems/helpers.py` (`MODELS`), and the Croissant file is built in
330
+ `croissantminer/croissant.py`.
331
+ - **A new system on the benchmark:** `make evaluate OUTPUTS=folder NAME=name` scores it with the paper's scorer and
332
+ judge model; the [leaderboard](https://github.com/berkearda/croissantminer/blob/main/leaderboard/README.md) explains the format and how to add your entry.
333
+ - **A wrong extraction, a bug or an idea:** open an [issue](https://github.com/berkearda/croissantminer/issues/new/choose)
334
+ or a [discussion](https://github.com/berkearda/croissantminer/discussions).
335
+
336
+ ## Repository layout
337
+
338
+ | Path | Contents |
339
+ |---|---|
340
+ | `croissantminer/` | Package: command line and Python API (`cli.py`, `api.py`), the six methods (`methods.py`) and the systems' code (`systems/`), the Croissant file (`croissant.py`), the extraction prompt, PDF reading and the ReAct agent |
341
+ | `scripts/` | Benchmark runs of the systems, the judge, tables and figures (guide in `scripts/README.md`) |
342
+ | `evaluation/` | Field metrics and system registry used by the scorer |
343
+ | `data/` | Gold annotations, judge verdicts and system outputs |
344
+ | `silver/` | Selection and extraction of the 500 silver papers |
345
+ | `tests/` | Tests and the published numbers they check against |
346
+ | `hf_space/` | The Hugging Face Space demo |
347
+ | `docs/` | Reproduction guide (`reproducing.md`), README figures and the ReAct agent's prompts |
348
+ | `legacy/` | Early prototype and experiments from before the paper, kept for reference and not maintained |
349
+
350
+ The scripts that read the named annotation sheets are not included, to protect the annotators'
351
+ privacy; `data/annotations/gold.parquet` and `data/annotations/iaa.parquet` are their output.
352
+ Comments that cite `decisions.md` or task numbers (`T-###`) refer to our
353
+ internal project log, which is not included.
354
+
355
+ ## Community
356
+
357
+ Questions and ideas go to [Discussions](https://github.com/berkearda/croissantminer/discussions), bugs and wrong
358
+ extractions to [issues](https://github.com/berkearda/croissantminer/issues). Please read
359
+ [CONTRIBUTING.md](https://github.com/berkearda/croissantminer/blob/main/CONTRIBUTING.md) before opening a pull request. Everyone taking part follows the
360
+ [code of conduct](https://github.com/berkearda/croissantminer/blob/main/CODE_OF_CONDUCT.md); security problems are reported as described in [SECURITY.md](https://github.com/berkearda/croissantminer/blob/main/SECURITY.md).
361
+ Changes are listed in [CHANGELOG.md](https://github.com/berkearda/croissantminer/blob/main/CHANGELOG.md).
362
+
363
+ ## Citation
364
+
365
+ ```bibtex
366
+ @inproceedings{arda2026croissantminer,
367
+ title = {CroissantMiner: Automated Extraction and Validation of Croissant Metadata for ML Datasets},
368
+ author = {Arda, Berke and Yavuz, Ahmetcan and Gerry, Paul and Lobentanzer, Sebastian and
369
+ Sarwar, Nobin and Giner-Miguelez, Joan and Chen, Kongtao and Zhang, Luyao and
370
+ Sachan, Mrinmaya and Akhtar, Mubashara},
371
+ booktitle = {Advances in Neural Information Processing Systems (Evaluations and Datasets Track)},
372
+ year = {2026}
373
+ }
374
+ ```
375
+
376
+ ## License
377
+
378
+ Code: MIT (see `LICENSE`). Annotations: CC BY 4.0 (see the dataset card). The papers remain under their
379
+ authors' licenses.