aied-unplugged 0.1.0__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- aied_unplugged-0.1.0/.gitignore +10 -0
- aied_unplugged-0.1.0/LICENSE +21 -0
- aied_unplugged-0.1.0/PKG-INFO +258 -0
- aied_unplugged-0.1.0/README.md +221 -0
- aied_unplugged-0.1.0/pyproject.toml +52 -0
- aied_unplugged-0.1.0/src/aied_unplugged/__init__.py +52 -0
- aied_unplugged-0.1.0/src/aied_unplugged/data.py +159 -0
- aied_unplugged-0.1.0/src/aied_unplugged/explore.py +192 -0
- aied_unplugged-0.1.0/src/aied_unplugged/graders/__init__.py +38 -0
- aied_unplugged-0.1.0/src/aied_unplugged/graders/answer_sheets.py +131 -0
- aied_unplugged-0.1.0/src/aied_unplugged/graders/common.py +113 -0
- aied_unplugged-0.1.0/src/aied_unplugged/graders/essays.py +91 -0
- aied_unplugged-0.1.0/src/aied_unplugged/graders/math_diagnostic.py +88 -0
- aied_unplugged-0.1.0/src/aied_unplugged/graders/metrics.py +70 -0
- aied_unplugged-0.1.0/src/aied_unplugged/models.py +256 -0
- aied_unplugged-0.1.0/src/aied_unplugged/submission.py +318 -0
- aied_unplugged-0.1.0/src/aied_unplugged/tracks.py +109 -0
- aied_unplugged-0.1.0/tests/test_graders.py +246 -0
- aied_unplugged-0.1.0/tests/test_submission_and_data.py +143 -0
- aied_unplugged-0.1.0/uv.lock +3742 -0
|
@@ -0,0 +1,21 @@
|
|
|
1
|
+
MIT License
|
|
2
|
+
|
|
3
|
+
Copyright (c) 2026 CESAR, UFRPE and AiBox Lab
|
|
4
|
+
|
|
5
|
+
Permission is hereby granted, free of charge, to any person obtaining a copy
|
|
6
|
+
of this software and associated documentation files (the "Software"), to deal
|
|
7
|
+
in the Software without restriction, including without limitation the rights
|
|
8
|
+
to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
|
|
9
|
+
copies of the Software, and to permit persons to whom the Software is
|
|
10
|
+
furnished to do so, subject to the following conditions:
|
|
11
|
+
|
|
12
|
+
The above copyright notice and this permission notice shall be included in all
|
|
13
|
+
copies or substantial portions of the Software.
|
|
14
|
+
|
|
15
|
+
THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
|
|
16
|
+
IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
|
|
17
|
+
FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
|
|
18
|
+
AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
|
|
19
|
+
LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
|
|
20
|
+
OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
|
|
21
|
+
SOFTWARE.
|
|
@@ -0,0 +1,258 @@
|
|
|
1
|
+
Metadata-Version: 2.5
|
|
2
|
+
Name: aied-unplugged
|
|
3
|
+
Version: 0.1.0
|
|
4
|
+
Summary: Loaders, graders and baselines for the AIED-Unplugged preview competition
|
|
5
|
+
Project-URL: Homepage, https://tools-competition.org/winner/aied/
|
|
6
|
+
Project-URL: Dataset, https://huggingface.co/datasets/aiboxlab/aied-unplugged-preview
|
|
7
|
+
License-Expression: MIT
|
|
8
|
+
License-File: LICENSE
|
|
9
|
+
Keywords: automated-essay-scoring,competition,education,handwriting,ocr
|
|
10
|
+
Classifier: Development Status :: 4 - Beta
|
|
11
|
+
Classifier: Intended Audience :: Science/Research
|
|
12
|
+
Classifier: Programming Language :: Python :: 3
|
|
13
|
+
Classifier: Topic :: Scientific/Engineering :: Artificial Intelligence
|
|
14
|
+
Requires-Python: >=3.10
|
|
15
|
+
Requires-Dist: numpy>=1.24
|
|
16
|
+
Requires-Dist: pandas>=2.0
|
|
17
|
+
Requires-Dist: pillow>=10
|
|
18
|
+
Requires-Dist: pyarrow>=14
|
|
19
|
+
Provides-Extra: all
|
|
20
|
+
Requires-Dist: accelerate>=0.30; extra == 'all'
|
|
21
|
+
Requires-Dist: datasets>=2.20; extra == 'all'
|
|
22
|
+
Requires-Dist: kagglehub>=0.3; extra == 'all'
|
|
23
|
+
Requires-Dist: matplotlib>=3.7; extra == 'all'
|
|
24
|
+
Requires-Dist: torch>=2.2; extra == 'all'
|
|
25
|
+
Requires-Dist: transformers>=4.44; extra == 'all'
|
|
26
|
+
Provides-Extra: explore
|
|
27
|
+
Requires-Dist: matplotlib>=3.7; extra == 'explore'
|
|
28
|
+
Provides-Extra: hf
|
|
29
|
+
Requires-Dist: datasets>=2.20; extra == 'hf'
|
|
30
|
+
Provides-Extra: kaggle
|
|
31
|
+
Requires-Dist: kagglehub>=0.3; extra == 'kaggle'
|
|
32
|
+
Provides-Extra: models
|
|
33
|
+
Requires-Dist: accelerate>=0.30; extra == 'models'
|
|
34
|
+
Requires-Dist: torch>=2.2; extra == 'models'
|
|
35
|
+
Requires-Dist: transformers>=4.44; extra == 'models'
|
|
36
|
+
Description-Content-Type: text/markdown
|
|
37
|
+
|
|
38
|
+
# AIED-Unplugged Dataset SDK
|
|
39
|
+
|
|
40
|
+
Python SDK for the [AIED Preview Competition](https://tools-competition.org/winner/aied/):
|
|
41
|
+
the official graders, dataset loaders, exploration helpers, and scaffolding to run a
|
|
42
|
+
Hugging Face model on a track.
|
|
43
|
+
|
|
44
|
+
```sh
|
|
45
|
+
pip install aied-unplugged # graders, loaders, submission tooling
|
|
46
|
+
pip install 'aied-unplugged[all]' # + kagglehub, datasets, matplotlib, transformers
|
|
47
|
+
```
|
|
48
|
+
|
|
49
|
+
The SDK is optional. It reads the published dataset and writes the published
|
|
50
|
+
submission format.
|
|
51
|
+
|
|
52
|
+
---
|
|
53
|
+
|
|
54
|
+
## The tracks
|
|
55
|
+
|
|
56
|
+
| Track | `track` | Predict | Ranked by |
|
|
57
|
+
|---|---|---|---|
|
|
58
|
+
| Automated Essay Scoring | `aes`, `essays` | `competence_1` … `competence_5` | mean QWK |
|
|
59
|
+
| Mathematical Exam Diagnostic | `math-diagnostic`, `math` | `diagnostic` | macro F1 |
|
|
60
|
+
| Answer Sheet Detection | `answer-sheet`, `sheets` | `answers` | cell accuracy |
|
|
61
|
+
|
|
62
|
+
---
|
|
63
|
+
|
|
64
|
+
## 1. Grading
|
|
65
|
+
|
|
66
|
+
The graders here are the ones the leaderboard runs, so your local score matches
|
|
67
|
+
your submitted one.
|
|
68
|
+
|
|
69
|
+
```python
|
|
70
|
+
from aied_unplugged import evaluate
|
|
71
|
+
|
|
72
|
+
result = evaluate("math", "my_predictions.csv", "validation_reference.csv")
|
|
73
|
+
result.primary # 0.41… the ranking metric
|
|
74
|
+
result.metrics # every metric the leaderboard records
|
|
75
|
+
result.per_example # one row per item: true, pred, correct
|
|
76
|
+
```
|
|
77
|
+
|
|
78
|
+
References can be a CSV or a `validation` split loaded from the dataset, with the
|
|
79
|
+
same target columns either way.
|
|
80
|
+
|
|
81
|
+
```python
|
|
82
|
+
from aied_unplugged import load_track, evaluate
|
|
83
|
+
|
|
84
|
+
validation = load_track("math", "validation")
|
|
85
|
+
evaluate("math", predictions, validation)
|
|
86
|
+
```
|
|
87
|
+
|
|
88
|
+
### Metrics
|
|
89
|
+
|
|
90
|
+
| Track | Primary | Definition | Also reported |
|
|
91
|
+
|---|---|---|---|
|
|
92
|
+
| Essays | mean QWK | Quadratic weighted kappa per competence, averaged over the five. Uses the full six-point scale rather than the values in the split, so scores don't shift with the split. | `exact_match` (per competence cell), `rmse` pooled over the five, `qwk_competence_1`…`_5`, `rmse_competence_1`…`_5` |
|
|
93
|
+
| Math | macro F1 | Unweighted mean of per-class F1 over the classes present in the reference. Predicting an absent class still costs you a false negative on the correct class. | `accuracy` |
|
|
94
|
+
| Answer sheets | cell accuracy | Correct cells over total reference cells, so a 26-question sheet counts more than a 16-question one. An omitted question counts as wrong. | `sheet_exact_match` (1 only when every question on a sheet matches), cell-level `macro_f1` |
|
|
95
|
+
|
|
96
|
+
A constant prediction zeroes the QWK expected-agreement denominator. That case
|
|
97
|
+
scores 1.0 when the prediction is right everywhere and 0.0 otherwise, instead of
|
|
98
|
+
NaN. Macro F1 skips classes absent from the reference: averaging over the full
|
|
99
|
+
taxonomy hands you a guaranteed zero for classes the split never uses, and the
|
|
100
|
+
preview sample omits two of the thirteen.
|
|
101
|
+
|
|
102
|
+
### Validation checks
|
|
103
|
+
|
|
104
|
+
A grader rejects the whole file instead of scoring the rows it can read. It raises
|
|
105
|
+
`SubmissionError` naming the ids and columns at fault, without quoting a target
|
|
106
|
+
value. Rejected files include:
|
|
107
|
+
|
|
108
|
+
- a missing column
|
|
109
|
+
- a blank or duplicated identifier
|
|
110
|
+
- an empty prediction
|
|
111
|
+
- an ungraded row, or a missing row
|
|
112
|
+
- a competence score off the 0/40/80/120/160/200 grid
|
|
113
|
+
- a diagnostic outside the thirteen labels
|
|
114
|
+
- an answer-sheet value outside the seven
|
|
115
|
+
- a not integer `question_number`
|
|
116
|
+
- a question repeated within a sheet
|
|
117
|
+
|
|
118
|
+
---
|
|
119
|
+
|
|
120
|
+
## 2. Loading
|
|
121
|
+
|
|
122
|
+
### Getting the dataset
|
|
123
|
+
|
|
124
|
+
`download()` pulls the dataset from Kaggle. Needs `aied-unplugged[kaggle]`.
|
|
125
|
+
|
|
126
|
+
```python
|
|
127
|
+
from aied_unplugged import download, load_track
|
|
128
|
+
|
|
129
|
+
download() # into the kagglehub cache, returns the path
|
|
130
|
+
load_track("math") # the loaders below now find it
|
|
131
|
+
load_track("math", source="kaggle") # same, in one call
|
|
132
|
+
load_track("math", source="hf") # in-memory DatasetDict from the HF Hub
|
|
133
|
+
```
|
|
134
|
+
|
|
135
|
+
Later calls reuse the cache; pass `force=True` to re-fetch. `download()` also sets
|
|
136
|
+
the default root for the process, so `explore` and `verify` run without `root=`.
|
|
137
|
+
|
|
138
|
+
Both mirrors carry the same release:
|
|
139
|
+
|
|
140
|
+
- Kaggle: <https://www.kaggle.com/datasets/aibox-lab/aied-unplugged-preview>
|
|
141
|
+
- Hugging Face: <https://huggingface.co/datasets/aiboxlab/aied-unplugged-preview>
|
|
142
|
+
|
|
143
|
+
To unpack one by hand, put `metadata/` and `schema/` under a single directory, then
|
|
144
|
+
name that directory with `root=`, with `AIED_UNPLUGGED_DATA`, or as
|
|
145
|
+
`competition-dataset/` in your working directory. `root=` wins over
|
|
146
|
+
`AIED_UNPLUGGED_DATA`, which wins over `download()`, which wins over
|
|
147
|
+
`competition-dataset/`.
|
|
148
|
+
|
|
149
|
+
### Loading a track
|
|
150
|
+
|
|
151
|
+
```python
|
|
152
|
+
from aied_unplugged import load_track, load_metadata, open_image
|
|
153
|
+
|
|
154
|
+
splits = load_track("essays") # local release tree if present, else the Hub
|
|
155
|
+
splits.train, splits.validation, splits.test
|
|
156
|
+
|
|
157
|
+
train = load_track("math", "train") # one split
|
|
158
|
+
image = open_image(train.iloc[0], "equation_image")
|
|
159
|
+
```
|
|
160
|
+
|
|
161
|
+
The local loaders return pandas DataFrames with the Parquet metadata and absolute
|
|
162
|
+
image paths in `<column>_path`.
|
|
163
|
+
|
|
164
|
+
```python
|
|
165
|
+
load_schema("math") # the published taxonomy, fields and metrics
|
|
166
|
+
graded_ids("aes") # the ids a submission must carry
|
|
167
|
+
sample_submission("aes") # the published placeholder file
|
|
168
|
+
verify() # re-check every SHA-256 in the release
|
|
169
|
+
```
|
|
170
|
+
|
|
171
|
+
---
|
|
172
|
+
|
|
173
|
+
## 3. Exploring
|
|
174
|
+
|
|
175
|
+
Needs `aied-unplugged[explore]` and a local copy of the dataset, from `download()`
|
|
176
|
+
or from a mirror you unpacked yourself.
|
|
177
|
+
|
|
178
|
+
```python
|
|
179
|
+
from aied_unplugged import explore
|
|
180
|
+
|
|
181
|
+
explore.summary() # rows per track and split
|
|
182
|
+
explore.label_distribution("math") # counts and shares, taxonomy order
|
|
183
|
+
explore.plot_label_distribution("math")
|
|
184
|
+
explore.describe_taxonomy("math") # labels with their Portuguese source terms
|
|
185
|
+
|
|
186
|
+
explore.show_essay("essay-1-3") # the page, with its five competence scores
|
|
187
|
+
explore.show_math("math-604-01") # question above, student working below
|
|
188
|
+
explore.show_sheet(0) # a sheet with what was marked
|
|
189
|
+
explore.render_math("math-604-01") # the same pair as one PIL image
|
|
190
|
+
```
|
|
191
|
+
|
|
192
|
+
Name an item by id, by position, or by passing its metadata row.
|
|
193
|
+
|
|
194
|
+
---
|
|
195
|
+
|
|
196
|
+
## 4. Running a model
|
|
197
|
+
|
|
198
|
+
Needs `aied-unplugged[models]`. Two strategies, either one enough for a valid
|
|
199
|
+
submission and a baseline score.
|
|
200
|
+
|
|
201
|
+
```python
|
|
202
|
+
from aied_unplugged import models, submission
|
|
203
|
+
|
|
204
|
+
# Predict the training majority for everything.
|
|
205
|
+
predictions = models.majority_baseline("math")
|
|
206
|
+
|
|
207
|
+
# Zero-shot with a local vision-language model.
|
|
208
|
+
predictions = models.predict("math", "Qwen/Qwen2.5-VL-3B-Instruct", limit=20)
|
|
209
|
+
|
|
210
|
+
submission.write("math", predictions, "submission.csv")
|
|
211
|
+
```
|
|
212
|
+
|
|
213
|
+
`predict` prompts the model once per item with the track's images and parses the
|
|
214
|
+
reply into the submission schema. An unparseable reply falls back to a safe
|
|
215
|
+
default, so a run always ends with a gradeable file.
|
|
216
|
+
|
|
217
|
+
Fine-tuning covers the math track only, as image classification over the student's
|
|
218
|
+
working:
|
|
219
|
+
|
|
220
|
+
```python
|
|
221
|
+
trainer = models.finetune(model="google/vit-base-patch16-224-in21k", epochs=3)
|
|
222
|
+
predictions = models.predict_with_classifier(trainer, split="test")
|
|
223
|
+
```
|
|
224
|
+
|
|
225
|
+
Not included: essay or answer-sheet fine-tuning, multi-GPU, LoRA, hyperparameter
|
|
226
|
+
search.
|
|
227
|
+
|
|
228
|
+
---
|
|
229
|
+
|
|
230
|
+
## 5. Submissions
|
|
231
|
+
|
|
232
|
+
```python
|
|
233
|
+
from aied_unplugged import build, validate, write, graded_ids
|
|
234
|
+
|
|
235
|
+
frame = build("answer-sheet", [{"sheet_id": "s1", "answers": {1: "A", 2: "Blank"}}])
|
|
236
|
+
report = validate(frame, "answer-sheet", expected_ids=graded_ids("answer-sheet"))
|
|
237
|
+
if not report.valid:
|
|
238
|
+
print(report)
|
|
239
|
+
write("answer-sheet", frame, "submission.csv")
|
|
240
|
+
```
|
|
241
|
+
|
|
242
|
+
`validate` runs the checks the competition site runs before upload, including the
|
|
243
|
+
JSON encoding of the answer-sheet column, plus a competence-scale check the site
|
|
244
|
+
does not yet perform. `build` takes answers as a list of records or as a
|
|
245
|
+
`{question_number: label}` mapping and encodes both to the same JSON.
|
|
246
|
+
|
|
247
|
+
---
|
|
248
|
+
|
|
249
|
+
## Development
|
|
250
|
+
|
|
251
|
+
```sh
|
|
252
|
+
uv run --with-editable . --with pytest pytest -q
|
|
253
|
+
uv run ruff check .
|
|
254
|
+
```
|
|
255
|
+
|
|
256
|
+
Tests needing the release tree skip when `competition-dataset/` is absent.
|
|
257
|
+
|
|
258
|
+
MIT licensed. The dataset itself is CC BY 4.0 and is published separately.
|
|
@@ -0,0 +1,221 @@
|
|
|
1
|
+
# AIED-Unplugged Dataset SDK
|
|
2
|
+
|
|
3
|
+
Python SDK for the [AIED Preview Competition](https://tools-competition.org/winner/aied/):
|
|
4
|
+
the official graders, dataset loaders, exploration helpers, and scaffolding to run a
|
|
5
|
+
Hugging Face model on a track.
|
|
6
|
+
|
|
7
|
+
```sh
|
|
8
|
+
pip install aied-unplugged # graders, loaders, submission tooling
|
|
9
|
+
pip install 'aied-unplugged[all]' # + kagglehub, datasets, matplotlib, transformers
|
|
10
|
+
```
|
|
11
|
+
|
|
12
|
+
The SDK is optional. It reads the published dataset and writes the published
|
|
13
|
+
submission format.
|
|
14
|
+
|
|
15
|
+
---
|
|
16
|
+
|
|
17
|
+
## The tracks
|
|
18
|
+
|
|
19
|
+
| Track | `track` | Predict | Ranked by |
|
|
20
|
+
|---|---|---|---|
|
|
21
|
+
| Automated Essay Scoring | `aes`, `essays` | `competence_1` … `competence_5` | mean QWK |
|
|
22
|
+
| Mathematical Exam Diagnostic | `math-diagnostic`, `math` | `diagnostic` | macro F1 |
|
|
23
|
+
| Answer Sheet Detection | `answer-sheet`, `sheets` | `answers` | cell accuracy |
|
|
24
|
+
|
|
25
|
+
---
|
|
26
|
+
|
|
27
|
+
## 1. Grading
|
|
28
|
+
|
|
29
|
+
The graders here are the ones the leaderboard runs, so your local score matches
|
|
30
|
+
your submitted one.
|
|
31
|
+
|
|
32
|
+
```python
|
|
33
|
+
from aied_unplugged import evaluate
|
|
34
|
+
|
|
35
|
+
result = evaluate("math", "my_predictions.csv", "validation_reference.csv")
|
|
36
|
+
result.primary # 0.41… the ranking metric
|
|
37
|
+
result.metrics # every metric the leaderboard records
|
|
38
|
+
result.per_example # one row per item: true, pred, correct
|
|
39
|
+
```
|
|
40
|
+
|
|
41
|
+
References can be a CSV or a `validation` split loaded from the dataset, with the
|
|
42
|
+
same target columns either way.
|
|
43
|
+
|
|
44
|
+
```python
|
|
45
|
+
from aied_unplugged import load_track, evaluate
|
|
46
|
+
|
|
47
|
+
validation = load_track("math", "validation")
|
|
48
|
+
evaluate("math", predictions, validation)
|
|
49
|
+
```
|
|
50
|
+
|
|
51
|
+
### Metrics
|
|
52
|
+
|
|
53
|
+
| Track | Primary | Definition | Also reported |
|
|
54
|
+
|---|---|---|---|
|
|
55
|
+
| Essays | mean QWK | Quadratic weighted kappa per competence, averaged over the five. Uses the full six-point scale rather than the values in the split, so scores don't shift with the split. | `exact_match` (per competence cell), `rmse` pooled over the five, `qwk_competence_1`…`_5`, `rmse_competence_1`…`_5` |
|
|
56
|
+
| Math | macro F1 | Unweighted mean of per-class F1 over the classes present in the reference. Predicting an absent class still costs you a false negative on the correct class. | `accuracy` |
|
|
57
|
+
| Answer sheets | cell accuracy | Correct cells over total reference cells, so a 26-question sheet counts more than a 16-question one. An omitted question counts as wrong. | `sheet_exact_match` (1 only when every question on a sheet matches), cell-level `macro_f1` |
|
|
58
|
+
|
|
59
|
+
A constant prediction zeroes the QWK expected-agreement denominator. That case
|
|
60
|
+
scores 1.0 when the prediction is right everywhere and 0.0 otherwise, instead of
|
|
61
|
+
NaN. Macro F1 skips classes absent from the reference: averaging over the full
|
|
62
|
+
taxonomy hands you a guaranteed zero for classes the split never uses, and the
|
|
63
|
+
preview sample omits two of the thirteen.
|
|
64
|
+
|
|
65
|
+
### Validation checks
|
|
66
|
+
|
|
67
|
+
A grader rejects the whole file instead of scoring the rows it can read. It raises
|
|
68
|
+
`SubmissionError` naming the ids and columns at fault, without quoting a target
|
|
69
|
+
value. Rejected files include:
|
|
70
|
+
|
|
71
|
+
- a missing column
|
|
72
|
+
- a blank or duplicated identifier
|
|
73
|
+
- an empty prediction
|
|
74
|
+
- an ungraded row, or a missing row
|
|
75
|
+
- a competence score off the 0/40/80/120/160/200 grid
|
|
76
|
+
- a diagnostic outside the thirteen labels
|
|
77
|
+
- an answer-sheet value outside the seven
|
|
78
|
+
- a not integer `question_number`
|
|
79
|
+
- a question repeated within a sheet
|
|
80
|
+
|
|
81
|
+
---
|
|
82
|
+
|
|
83
|
+
## 2. Loading
|
|
84
|
+
|
|
85
|
+
### Getting the dataset
|
|
86
|
+
|
|
87
|
+
`download()` pulls the dataset from Kaggle. Needs `aied-unplugged[kaggle]`.
|
|
88
|
+
|
|
89
|
+
```python
|
|
90
|
+
from aied_unplugged import download, load_track
|
|
91
|
+
|
|
92
|
+
download() # into the kagglehub cache, returns the path
|
|
93
|
+
load_track("math") # the loaders below now find it
|
|
94
|
+
load_track("math", source="kaggle") # same, in one call
|
|
95
|
+
load_track("math", source="hf") # in-memory DatasetDict from the HF Hub
|
|
96
|
+
```
|
|
97
|
+
|
|
98
|
+
Later calls reuse the cache; pass `force=True` to re-fetch. `download()` also sets
|
|
99
|
+
the default root for the process, so `explore` and `verify` run without `root=`.
|
|
100
|
+
|
|
101
|
+
Both mirrors carry the same release:
|
|
102
|
+
|
|
103
|
+
- Kaggle: <https://www.kaggle.com/datasets/aibox-lab/aied-unplugged-preview>
|
|
104
|
+
- Hugging Face: <https://huggingface.co/datasets/aiboxlab/aied-unplugged-preview>
|
|
105
|
+
|
|
106
|
+
To unpack one by hand, put `metadata/` and `schema/` under a single directory, then
|
|
107
|
+
name that directory with `root=`, with `AIED_UNPLUGGED_DATA`, or as
|
|
108
|
+
`competition-dataset/` in your working directory. `root=` wins over
|
|
109
|
+
`AIED_UNPLUGGED_DATA`, which wins over `download()`, which wins over
|
|
110
|
+
`competition-dataset/`.
|
|
111
|
+
|
|
112
|
+
### Loading a track
|
|
113
|
+
|
|
114
|
+
```python
|
|
115
|
+
from aied_unplugged import load_track, load_metadata, open_image
|
|
116
|
+
|
|
117
|
+
splits = load_track("essays") # local release tree if present, else the Hub
|
|
118
|
+
splits.train, splits.validation, splits.test
|
|
119
|
+
|
|
120
|
+
train = load_track("math", "train") # one split
|
|
121
|
+
image = open_image(train.iloc[0], "equation_image")
|
|
122
|
+
```
|
|
123
|
+
|
|
124
|
+
The local loaders return pandas DataFrames with the Parquet metadata and absolute
|
|
125
|
+
image paths in `<column>_path`.
|
|
126
|
+
|
|
127
|
+
```python
|
|
128
|
+
load_schema("math") # the published taxonomy, fields and metrics
|
|
129
|
+
graded_ids("aes") # the ids a submission must carry
|
|
130
|
+
sample_submission("aes") # the published placeholder file
|
|
131
|
+
verify() # re-check every SHA-256 in the release
|
|
132
|
+
```
|
|
133
|
+
|
|
134
|
+
---
|
|
135
|
+
|
|
136
|
+
## 3. Exploring
|
|
137
|
+
|
|
138
|
+
Needs `aied-unplugged[explore]` and a local copy of the dataset, from `download()`
|
|
139
|
+
or from a mirror you unpacked yourself.
|
|
140
|
+
|
|
141
|
+
```python
|
|
142
|
+
from aied_unplugged import explore
|
|
143
|
+
|
|
144
|
+
explore.summary() # rows per track and split
|
|
145
|
+
explore.label_distribution("math") # counts and shares, taxonomy order
|
|
146
|
+
explore.plot_label_distribution("math")
|
|
147
|
+
explore.describe_taxonomy("math") # labels with their Portuguese source terms
|
|
148
|
+
|
|
149
|
+
explore.show_essay("essay-1-3") # the page, with its five competence scores
|
|
150
|
+
explore.show_math("math-604-01") # question above, student working below
|
|
151
|
+
explore.show_sheet(0) # a sheet with what was marked
|
|
152
|
+
explore.render_math("math-604-01") # the same pair as one PIL image
|
|
153
|
+
```
|
|
154
|
+
|
|
155
|
+
Name an item by id, by position, or by passing its metadata row.
|
|
156
|
+
|
|
157
|
+
---
|
|
158
|
+
|
|
159
|
+
## 4. Running a model
|
|
160
|
+
|
|
161
|
+
Needs `aied-unplugged[models]`. Two strategies, either one enough for a valid
|
|
162
|
+
submission and a baseline score.
|
|
163
|
+
|
|
164
|
+
```python
|
|
165
|
+
from aied_unplugged import models, submission
|
|
166
|
+
|
|
167
|
+
# Predict the training majority for everything.
|
|
168
|
+
predictions = models.majority_baseline("math")
|
|
169
|
+
|
|
170
|
+
# Zero-shot with a local vision-language model.
|
|
171
|
+
predictions = models.predict("math", "Qwen/Qwen2.5-VL-3B-Instruct", limit=20)
|
|
172
|
+
|
|
173
|
+
submission.write("math", predictions, "submission.csv")
|
|
174
|
+
```
|
|
175
|
+
|
|
176
|
+
`predict` prompts the model once per item with the track's images and parses the
|
|
177
|
+
reply into the submission schema. An unparseable reply falls back to a safe
|
|
178
|
+
default, so a run always ends with a gradeable file.
|
|
179
|
+
|
|
180
|
+
Fine-tuning covers the math track only, as image classification over the student's
|
|
181
|
+
working:
|
|
182
|
+
|
|
183
|
+
```python
|
|
184
|
+
trainer = models.finetune(model="google/vit-base-patch16-224-in21k", epochs=3)
|
|
185
|
+
predictions = models.predict_with_classifier(trainer, split="test")
|
|
186
|
+
```
|
|
187
|
+
|
|
188
|
+
Not included: essay or answer-sheet fine-tuning, multi-GPU, LoRA, hyperparameter
|
|
189
|
+
search.
|
|
190
|
+
|
|
191
|
+
---
|
|
192
|
+
|
|
193
|
+
## 5. Submissions
|
|
194
|
+
|
|
195
|
+
```python
|
|
196
|
+
from aied_unplugged import build, validate, write, graded_ids
|
|
197
|
+
|
|
198
|
+
frame = build("answer-sheet", [{"sheet_id": "s1", "answers": {1: "A", 2: "Blank"}}])
|
|
199
|
+
report = validate(frame, "answer-sheet", expected_ids=graded_ids("answer-sheet"))
|
|
200
|
+
if not report.valid:
|
|
201
|
+
print(report)
|
|
202
|
+
write("answer-sheet", frame, "submission.csv")
|
|
203
|
+
```
|
|
204
|
+
|
|
205
|
+
`validate` runs the checks the competition site runs before upload, including the
|
|
206
|
+
JSON encoding of the answer-sheet column, plus a competence-scale check the site
|
|
207
|
+
does not yet perform. `build` takes answers as a list of records or as a
|
|
208
|
+
`{question_number: label}` mapping and encodes both to the same JSON.
|
|
209
|
+
|
|
210
|
+
---
|
|
211
|
+
|
|
212
|
+
## Development
|
|
213
|
+
|
|
214
|
+
```sh
|
|
215
|
+
uv run --with-editable . --with pytest pytest -q
|
|
216
|
+
uv run ruff check .
|
|
217
|
+
```
|
|
218
|
+
|
|
219
|
+
Tests needing the release tree skip when `competition-dataset/` is absent.
|
|
220
|
+
|
|
221
|
+
MIT licensed. The dataset itself is CC BY 4.0 and is published separately.
|
|
@@ -0,0 +1,52 @@
|
|
|
1
|
+
[build-system]
|
|
2
|
+
requires = ["hatchling"]
|
|
3
|
+
build-backend = "hatchling.build"
|
|
4
|
+
|
|
5
|
+
[project]
|
|
6
|
+
name = "aied-unplugged"
|
|
7
|
+
version = "0.1.0"
|
|
8
|
+
description = "Loaders, graders and baselines for the AIED-Unplugged preview competition"
|
|
9
|
+
readme = "README.md"
|
|
10
|
+
requires-python = ">=3.10"
|
|
11
|
+
license = "MIT"
|
|
12
|
+
license-files = ["LICENSE"]
|
|
13
|
+
keywords = ["handwriting", "education", "competition", "automated-essay-scoring", "ocr"]
|
|
14
|
+
classifiers = [
|
|
15
|
+
"Development Status :: 4 - Beta",
|
|
16
|
+
"Intended Audience :: Science/Research",
|
|
17
|
+
"Programming Language :: Python :: 3",
|
|
18
|
+
"Topic :: Scientific/Engineering :: Artificial Intelligence",
|
|
19
|
+
]
|
|
20
|
+
dependencies = [
|
|
21
|
+
"numpy>=1.24",
|
|
22
|
+
"pandas>=2.0",
|
|
23
|
+
"pyarrow>=14",
|
|
24
|
+
"pillow>=10",
|
|
25
|
+
]
|
|
26
|
+
|
|
27
|
+
[project.optional-dependencies]
|
|
28
|
+
hf = ["datasets>=2.20"]
|
|
29
|
+
kaggle = ["kagglehub>=0.3"]
|
|
30
|
+
explore = ["matplotlib>=3.7"]
|
|
31
|
+
models = ["transformers>=4.44", "torch>=2.2", "accelerate>=0.30"]
|
|
32
|
+
all = ["aied-unplugged[hf,kaggle,explore,models]"]
|
|
33
|
+
|
|
34
|
+
[project.urls]
|
|
35
|
+
Homepage = "https://tools-competition.org/winner/aied/"
|
|
36
|
+
Dataset = "https://huggingface.co/datasets/aiboxlab/aied-unplugged-preview"
|
|
37
|
+
|
|
38
|
+
[dependency-groups]
|
|
39
|
+
dev = ["pytest>=8", "ruff>=0.6"]
|
|
40
|
+
|
|
41
|
+
[tool.hatch.build.targets.wheel]
|
|
42
|
+
packages = ["src/aied_unplugged"]
|
|
43
|
+
|
|
44
|
+
[tool.ruff]
|
|
45
|
+
line-length = 96
|
|
46
|
+
src = ["src", "tests"]
|
|
47
|
+
|
|
48
|
+
[tool.ruff.lint]
|
|
49
|
+
select = ["E", "F", "I", "UP", "B", "SIM"]
|
|
50
|
+
|
|
51
|
+
[tool.pytest.ini_options]
|
|
52
|
+
testpaths = ["tests"]
|
|
@@ -0,0 +1,52 @@
|
|
|
1
|
+
from __future__ import annotations
|
|
2
|
+
|
|
3
|
+
from .data import (
|
|
4
|
+
download,
|
|
5
|
+
graded_ids,
|
|
6
|
+
load_metadata,
|
|
7
|
+
load_schema,
|
|
8
|
+
load_track,
|
|
9
|
+
open_image,
|
|
10
|
+
sample_submission,
|
|
11
|
+
verify,
|
|
12
|
+
)
|
|
13
|
+
from .graders import GraderResult, SubmissionError, evaluate, get_grader
|
|
14
|
+
from .submission import ValidationReport, build, validate, write
|
|
15
|
+
from .tracks import (
|
|
16
|
+
ANSWER_VALUES,
|
|
17
|
+
COMPETENCES,
|
|
18
|
+
DIAGNOSTIC_LABELS,
|
|
19
|
+
SCORE_VALUES,
|
|
20
|
+
TRACKS,
|
|
21
|
+
Track,
|
|
22
|
+
get_track,
|
|
23
|
+
)
|
|
24
|
+
|
|
25
|
+
__version__ = "0.1.0"
|
|
26
|
+
|
|
27
|
+
__all__ = [
|
|
28
|
+
"ANSWER_VALUES",
|
|
29
|
+
"COMPETENCES",
|
|
30
|
+
"DIAGNOSTIC_LABELS",
|
|
31
|
+
"SCORE_VALUES",
|
|
32
|
+
"TRACKS",
|
|
33
|
+
"GraderResult",
|
|
34
|
+
"SubmissionError",
|
|
35
|
+
"Track",
|
|
36
|
+
"ValidationReport",
|
|
37
|
+
"__version__",
|
|
38
|
+
"build",
|
|
39
|
+
"download",
|
|
40
|
+
"evaluate",
|
|
41
|
+
"get_grader",
|
|
42
|
+
"get_track",
|
|
43
|
+
"graded_ids",
|
|
44
|
+
"load_metadata",
|
|
45
|
+
"load_schema",
|
|
46
|
+
"load_track",
|
|
47
|
+
"open_image",
|
|
48
|
+
"sample_submission",
|
|
49
|
+
"validate",
|
|
50
|
+
"verify",
|
|
51
|
+
"write",
|
|
52
|
+
]
|