marin-taskcompendium 0.2.160.dev38062449553__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- marin_taskcompendium-0.2.160.dev38062449553/.gitignore +253 -0
- marin_taskcompendium-0.2.160.dev38062449553/PKG-INFO +16 -0
- marin_taskcompendium-0.2.160.dev38062449553/README.md +356 -0
- marin_taskcompendium-0.2.160.dev38062449553/pyproject.toml +58 -0
- marin_taskcompendium-0.2.160.dev38062449553/src/taskcompendium/__init__.py +2 -0
- marin_taskcompendium-0.2.160.dev38062449553/src/taskcompendium/chat.py +93 -0
- marin_taskcompendium-0.2.160.dev38062449553/src/taskcompendium/convert/__init__.py +4 -0
- marin_taskcompendium-0.2.160.dev38062449553/src/taskcompendium/convert/answers.py +210 -0
- marin_taskcompendium-0.2.160.dev38062449553/src/taskcompendium/convert/code.py +78 -0
- marin_taskcompendium-0.2.160.dev38062449553/src/taskcompendium/convert/conversation.py +44 -0
- marin_taskcompendium-0.2.160.dev38062449553/src/taskcompendium/convert/delivery.py +34 -0
- marin_taskcompendium-0.2.160.dev38062449553/src/taskcompendium/convert/environment.py +16 -0
- marin_taskcompendium-0.2.160.dev38062449553/src/taskcompendium/convert/executable.py +348 -0
- marin_taskcompendium-0.2.160.dev38062449553/src/taskcompendium/convert/json_schema.py +32 -0
- marin_taskcompendium-0.2.160.dev38062449553/src/taskcompendium/convert/nemotron_ultra.py +225 -0
- marin_taskcompendium-0.2.160.dev38062449553/src/taskcompendium/convert/preference.py +87 -0
- marin_taskcompendium-0.2.160.dev38062449553/src/taskcompendium/convert/script_grader.py +61 -0
- marin_taskcompendium-0.2.160.dev38062449553/src/taskcompendium/convert/tasktrove.py +184 -0
- marin_taskcompendium-0.2.160.dev38062449553/src/taskcompendium/convert/tasktrove_code_contests.py +99 -0
- marin_taskcompendium-0.2.160.dev38062449553/src/taskcompendium/convert/tasktrove_codeforces.py +117 -0
- marin_taskcompendium-0.2.160.dev38062449553/src/taskcompendium/convert/tasktrove_converted_task.py +88 -0
- marin_taskcompendium-0.2.160.dev38062449553/src/taskcompendium/convert/tasktrove_json_schemas.py +133 -0
- marin_taskcompendium-0.2.160.dev38062449553/src/taskcompendium/convert/tasktrove_nemotron_data.py +25 -0
- marin_taskcompendium-0.2.160.dev38062449553/src/taskcompendium/convert/tasktrove_nemotron_structured_outputs.py +284 -0
- marin_taskcompendium-0.2.160.dev38062449553/src/taskcompendium/convert/tasktrove_nl2bash.py +168 -0
- marin_taskcompendium-0.2.160.dev38062449553/src/taskcompendium/convert/tasktrove_python_unit_tests.py +100 -0
- marin_taskcompendium-0.2.160.dev38062449553/src/taskcompendium/convert/tasktrove_stdio_cases.py +91 -0
- marin_taskcompendium-0.2.160.dev38062449553/src/taskcompendium/convert/tasktrove_taco.py +88 -0
- marin_taskcompendium-0.2.160.dev38062449553/src/taskcompendium/convert/verifyit_build.py +62 -0
- marin_taskcompendium-0.2.160.dev38062449553/src/taskcompendium/direct_chat.py +40 -0
- marin_taskcompendium-0.2.160.dev38062449553/src/taskcompendium/grader.py +42 -0
- marin_taskcompendium-0.2.160.dev38062449553/src/taskcompendium/grading.py +118 -0
- marin_taskcompendium-0.2.160.dev38062449553/src/taskcompendium/grading_result.py +37 -0
- marin_taskcompendium-0.2.160.dev38062449553/src/taskcompendium/harbor/__init__.py +4 -0
- marin_taskcompendium-0.2.160.dev38062449553/src/taskcompendium/harbor/compare.py +255 -0
- marin_taskcompendium-0.2.160.dev38062449553/src/taskcompendium/harbor/export.py +644 -0
- marin_taskcompendium-0.2.160.dev38062449553/src/taskcompendium/harbor/records.py +79 -0
- marin_taskcompendium-0.2.160.dev38062449553/src/taskcompendium/harbor/snapshots.py +122 -0
- marin_taskcompendium-0.2.160.dev38062449553/src/taskcompendium/harbor/tasktrove.py +72 -0
- marin_taskcompendium-0.2.160.dev38062449553/src/taskcompendium/importers/__init__.py +4 -0
- marin_taskcompendium-0.2.160.dev38062449553/src/taskcompendium/importers/nemo_predicted_action.py +200 -0
- marin_taskcompendium-0.2.160.dev38062449553/src/taskcompendium/importers/swe.py +78 -0
- marin_taskcompendium-0.2.160.dev38062449553/src/taskcompendium/importers/tasktrove/__init__.py +4 -0
- marin_taskcompendium-0.2.160.dev38062449553/src/taskcompendium/importers/tasktrove/convert.py +70 -0
- marin_taskcompendium-0.2.160.dev38062449553/src/taskcompendium/importers/tasktrove/mcqa.py +80 -0
- marin_taskcompendium-0.2.160.dev38062449553/src/taskcompendium/importers/tasktrove/models.py +30 -0
- marin_taskcompendium-0.2.160.dev38062449553/src/taskcompendium/models.py +891 -0
- marin_taskcompendium-0.2.160.dev38062449553/src/taskcompendium/pipeline/README.md +223 -0
- marin_taskcompendium-0.2.160.dev38062449553/src/taskcompendium/pipeline/__init__.py +4 -0
- marin_taskcompendium-0.2.160.dev38062449553/src/taskcompendium/pipeline/audit_schema.py +119 -0
- marin_taskcompendium-0.2.160.dev38062449553/src/taskcompendium/pipeline/chat_requests.py +100 -0
- marin_taskcompendium-0.2.160.dev38062449553/src/taskcompendium/pipeline/controls.py +270 -0
- marin_taskcompendium-0.2.160.dev38062449553/src/taskcompendium/pipeline/conversion.py +148 -0
- marin_taskcompendium-0.2.160.dev38062449553/src/taskcompendium/pipeline/environment_inventory.py +49 -0
- marin_taskcompendium-0.2.160.dev38062449553/src/taskcompendium/pipeline/execution_telemetry.py +141 -0
- marin_taskcompendium-0.2.160.dev38062449553/src/taskcompendium/pipeline/filtering.py +37 -0
- marin_taskcompendium-0.2.160.dev38062449553/src/taskcompendium/pipeline/fingerprints.py +110 -0
- marin_taskcompendium-0.2.160.dev38062449553/src/taskcompendium/pipeline/inputs.py +98 -0
- marin_taskcompendium-0.2.160.dev38062449553/src/taskcompendium/pipeline/models.py +325 -0
- marin_taskcompendium-0.2.160.dev38062449553/src/taskcompendium/pipeline/query_cache.py +259 -0
- marin_taskcompendium-0.2.160.dev38062449553/src/taskcompendium/pipeline/review.py +785 -0
- marin_taskcompendium-0.2.160.dev38062449553/src/taskcompendium/pipeline/review_requests.py +285 -0
- marin_taskcompendium-0.2.160.dev38062449553/src/taskcompendium/pipeline/sampling.py +44 -0
- marin_taskcompendium-0.2.160.dev38062449553/src/taskcompendium/pipeline/shard_outputs.py +58 -0
- marin_taskcompendium-0.2.160.dev38062449553/src/taskcompendium/pipeline/source_processing.py +1072 -0
- marin_taskcompendium-0.2.160.dev38062449553/src/taskcompendium/pipeline/source_quality.py +283 -0
- marin_taskcompendium-0.2.160.dev38062449553/src/taskcompendium/pipeline/source_verification.py +634 -0
- marin_taskcompendium-0.2.160.dev38062449553/src/taskcompendium/pipeline/sources.py +210 -0
- marin_taskcompendium-0.2.160.dev38062449553/src/taskcompendium/pipeline/stages.py +845 -0
- marin_taskcompendium-0.2.160.dev38062449553/src/taskcompendium/pipeline/transforms.py +221 -0
- marin_taskcompendium-0.2.160.dev38062449553/src/taskcompendium/pipeline/verification.py +192 -0
- marin_taskcompendium-0.2.160.dev38062449553/src/taskcompendium/runtime/__init__.py +7 -0
- marin_taskcompendium-0.2.160.dev38062449553/src/taskcompendium/runtime/environment.py +27 -0
- marin_taskcompendium-0.2.160.dev38062449553/src/taskcompendium/runtime/episode.py +98 -0
- marin_taskcompendium-0.2.160.dev38062449553/src/taskcompendium/runtime/grading.py +649 -0
- marin_taskcompendium-0.2.160.dev38062449553/src/taskcompendium/runtime/local.py +219 -0
- marin_taskcompendium-0.2.160.dev38062449553/src/taskcompendium/runtime/models.py +82 -0
- marin_taskcompendium-0.2.160.dev38062449553/src/taskcompendium/runtime/output_capture.py +113 -0
- marin_taskcompendium-0.2.160.dev38062449553/src/taskcompendium/runtime/resources.py +18 -0
- marin_taskcompendium-0.2.160.dev38062449553/src/taskcompendium/runtime/shell.py +286 -0
- marin_taskcompendium-0.2.160.dev38062449553/src/taskcompendium/runtime/task_grading.py +61 -0
- marin_taskcompendium-0.2.160.dev38062449553/src/taskcompendium/submission.py +188 -0
- marin_taskcompendium-0.2.160.dev38062449553/tests/conftest.py +13 -0
- marin_taskcompendium-0.2.160.dev38062449553/tests/fixtures/nemo/predicted-action.json +1 -0
- marin_taskcompendium-0.2.160.dev38062449553/tests/fixtures/nemo/predicted-action.provenance.json +11 -0
- marin_taskcompendium-0.2.160.dev38062449553/tests/fixtures/review/stack_pytest_long_evidence.json +12 -0
- marin_taskcompendium-0.2.160.dev38062449553/tests/fixtures/tasktrove/mcq-1961bdb52b5a.tar.gz +0 -0
- marin_taskcompendium-0.2.160.dev38062449553/tests/pipeline_stages.py +192 -0
- marin_taskcompendium-0.2.160.dev38062449553/tests/test_answer_formats.py +194 -0
- marin_taskcompendium-0.2.160.dev38062449553/tests/test_chat_requests.py +82 -0
- marin_taskcompendium-0.2.160.dev38062449553/tests/test_chat_review.py +199 -0
- marin_taskcompendium-0.2.160.dev38062449553/tests/test_controls.py +231 -0
- marin_taskcompendium-0.2.160.dev38062449553/tests/test_conversion.py +212 -0
- marin_taskcompendium-0.2.160.dev38062449553/tests/test_convert_answers.py +176 -0
- marin_taskcompendium-0.2.160.dev38062449553/tests/test_convert_executable.py +418 -0
- marin_taskcompendium-0.2.160.dev38062449553/tests/test_convert_json_schema.py +27 -0
- marin_taskcompendium-0.2.160.dev38062449553/tests/test_convert_nemotron_ultra.py +114 -0
- marin_taskcompendium-0.2.160.dev38062449553/tests/test_convert_preference.py +84 -0
- marin_taskcompendium-0.2.160.dev38062449553/tests/test_convert_script_grader.py +70 -0
- marin_taskcompendium-0.2.160.dev38062449553/tests/test_convert_tasktrove.py +89 -0
- marin_taskcompendium-0.2.160.dev38062449553/tests/test_environment.py +23 -0
- marin_taskcompendium-0.2.160.dev38062449553/tests/test_execution_telemetry.py +99 -0
- marin_taskcompendium-0.2.160.dev38062449553/tests/test_grader_package.py +206 -0
- marin_taskcompendium-0.2.160.dev38062449553/tests/test_harbor_compare.py +238 -0
- marin_taskcompendium-0.2.160.dev38062449553/tests/test_harbor_export.py +65 -0
- marin_taskcompendium-0.2.160.dev38062449553/tests/test_local_runtime.py +27 -0
- marin_taskcompendium-0.2.160.dev38062449553/tests/test_multiple_choice_verifier.py +47 -0
- marin_taskcompendium-0.2.160.dev38062449553/tests/test_nemo_predicted_action.py +319 -0
- marin_taskcompendium-0.2.160.dev38062449553/tests/test_output_capture.py +263 -0
- marin_taskcompendium-0.2.160.dev38062449553/tests/test_pipeline.py +1319 -0
- marin_taskcompendium-0.2.160.dev38062449553/tests/test_review_evidence.py +147 -0
- marin_taskcompendium-0.2.160.dev38062449553/tests/test_runtime.py +428 -0
- marin_taskcompendium-0.2.160.dev38062449553/tests/test_schema.py +318 -0
- marin_taskcompendium-0.2.160.dev38062449553/tests/test_script_grader.py +551 -0
- marin_taskcompendium-0.2.160.dev38062449553/tests/test_source_processing.py +897 -0
- marin_taskcompendium-0.2.160.dev38062449553/tests/test_source_quality.py +395 -0
- marin_taskcompendium-0.2.160.dev38062449553/tests/test_source_verification.py +560 -0
- marin_taskcompendium-0.2.160.dev38062449553/tests/test_submission_grading.py +191 -0
- marin_taskcompendium-0.2.160.dev38062449553/tests/test_tasktrove_answers.py +173 -0
|
@@ -0,0 +1,253 @@
|
|
|
1
|
+
# redundant, but Ray looks for this otherwise.
|
|
2
|
+
.git
|
|
3
|
+
|
|
4
|
+
logs/
|
|
5
|
+
|
|
6
|
+
# CPU profiles
|
|
7
|
+
prof/
|
|
8
|
+
|
|
9
|
+
# Downloaded build tools (zig, etc.)
|
|
10
|
+
.tools/
|
|
11
|
+
|
|
12
|
+
tests/snapshots/outputs
|
|
13
|
+
tests/snapshots/diffs
|
|
14
|
+
|
|
15
|
+
# don't log data/MD outputs to git
|
|
16
|
+
data/*
|
|
17
|
+
output/*
|
|
18
|
+
outputs/*
|
|
19
|
+
|
|
20
|
+
# Snapshot diffs and outputs
|
|
21
|
+
tests/snapshots/*/outputs/*
|
|
22
|
+
tests/snapshots/*/diffs/*
|
|
23
|
+
|
|
24
|
+
# This is mainly for Ray and using submodule
|
|
25
|
+
*/**/.git
|
|
26
|
+
|
|
27
|
+
### Python template
|
|
28
|
+
# Byte-compiled / optimized / DLL files
|
|
29
|
+
__pycache__/
|
|
30
|
+
*.py[cod]
|
|
31
|
+
*$py.class
|
|
32
|
+
|
|
33
|
+
# C extensions
|
|
34
|
+
*.so
|
|
35
|
+
|
|
36
|
+
# pypa/gh-action-pypi-publish caches its Docker action manifest here.
|
|
37
|
+
.github/.tmp/
|
|
38
|
+
|
|
39
|
+
# Distribution / packaging
|
|
40
|
+
.Python
|
|
41
|
+
build/
|
|
42
|
+
develop-eggs/
|
|
43
|
+
dist/
|
|
44
|
+
downloads/
|
|
45
|
+
eggs/
|
|
46
|
+
.eggs/
|
|
47
|
+
lib64/
|
|
48
|
+
parts/
|
|
49
|
+
sdist/
|
|
50
|
+
local_store/
|
|
51
|
+
wheels/
|
|
52
|
+
share/python-wheels/
|
|
53
|
+
*.egg-info/
|
|
54
|
+
.installed.cfg
|
|
55
|
+
*.egg
|
|
56
|
+
MANIFEST
|
|
57
|
+
|
|
58
|
+
# PyInstaller
|
|
59
|
+
# Usually these files are written by a python script from a template
|
|
60
|
+
# before PyInstaller builds the exe, so as to inject date/other infos into it.
|
|
61
|
+
*.manifest
|
|
62
|
+
*.spec
|
|
63
|
+
|
|
64
|
+
# Installer logs
|
|
65
|
+
pip-log.txt
|
|
66
|
+
pip-delete-this-directory.txt
|
|
67
|
+
|
|
68
|
+
# Unit test / coverage reports
|
|
69
|
+
htmlcov/
|
|
70
|
+
.tox/
|
|
71
|
+
.nox/
|
|
72
|
+
.coverage
|
|
73
|
+
.coverage.*
|
|
74
|
+
.cache
|
|
75
|
+
nosetests.xml
|
|
76
|
+
coverage.xml
|
|
77
|
+
*.cover
|
|
78
|
+
*.py,cover
|
|
79
|
+
.hypothesis/
|
|
80
|
+
.pytest_cache/
|
|
81
|
+
cover/
|
|
82
|
+
|
|
83
|
+
# Translations
|
|
84
|
+
*.mo
|
|
85
|
+
*.pot
|
|
86
|
+
|
|
87
|
+
# Django stuff:
|
|
88
|
+
*.log
|
|
89
|
+
local_settings.py
|
|
90
|
+
db.sqlite3
|
|
91
|
+
db.sqlite3-journal
|
|
92
|
+
|
|
93
|
+
# Flask stuff:
|
|
94
|
+
instance/
|
|
95
|
+
.webassets-cache
|
|
96
|
+
|
|
97
|
+
# Scrapy stuff:
|
|
98
|
+
.scrapy
|
|
99
|
+
|
|
100
|
+
# Sphinx documentation
|
|
101
|
+
docs/_build/
|
|
102
|
+
|
|
103
|
+
# PyBuilder
|
|
104
|
+
.pybuilder/
|
|
105
|
+
target/
|
|
106
|
+
|
|
107
|
+
# Jupyter Notebook
|
|
108
|
+
.ipynb_checkpoints
|
|
109
|
+
|
|
110
|
+
# IPython
|
|
111
|
+
profile_default/
|
|
112
|
+
ipython_config.py
|
|
113
|
+
|
|
114
|
+
# pyenv
|
|
115
|
+
# For a library or package, you might want to ignore these files since the code is
|
|
116
|
+
# intended to run in multiple environments; otherwise, check them in:
|
|
117
|
+
# .python-version
|
|
118
|
+
|
|
119
|
+
# pipenv
|
|
120
|
+
# According to pypa/pipenv#598, it is recommended to include Pipfile.lock in version control.
|
|
121
|
+
# However, in case of collaboration, if having platform-specific dependencies or dependencies
|
|
122
|
+
# having no cross-platform support, pipenv may install dependencies that don't work, or not
|
|
123
|
+
# install all needed dependencies.
|
|
124
|
+
#Pipfile.lock
|
|
125
|
+
|
|
126
|
+
# poetry
|
|
127
|
+
# Similar to Pipfile.lock, it is generally recommended to include poetry.lock in version control.
|
|
128
|
+
# This is especially recommended for binary packages to ensure reproducibility, and is more
|
|
129
|
+
# commonly ignored for libraries.
|
|
130
|
+
# https://python-poetry.org/docs/basic-usage/#commit-your-poetrylock-file-to-version-control
|
|
131
|
+
#poetry.lock
|
|
132
|
+
|
|
133
|
+
# pdm
|
|
134
|
+
# Similar to Pipfile.lock, it is generally recommended to include pdm.lock in version control.
|
|
135
|
+
#pdm.lock
|
|
136
|
+
# pdm stores project-wide configurations in .pdm.toml, but it is recommended to not include it
|
|
137
|
+
# in version control.
|
|
138
|
+
# https://pdm.fming.dev/#use-with-ide
|
|
139
|
+
.pdm.toml
|
|
140
|
+
|
|
141
|
+
# PEP 582; used by e.g. github.com/David-OConnor/pyflow and github.com/pdm-project/pdm
|
|
142
|
+
__pypackages__/
|
|
143
|
+
|
|
144
|
+
# Celery stuff
|
|
145
|
+
celerybeat-schedule
|
|
146
|
+
celerybeat.pid
|
|
147
|
+
|
|
148
|
+
# SageMath parsed files
|
|
149
|
+
*.sage.py
|
|
150
|
+
|
|
151
|
+
# Environments
|
|
152
|
+
.env
|
|
153
|
+
.venv
|
|
154
|
+
env/
|
|
155
|
+
venv/
|
|
156
|
+
ENV/
|
|
157
|
+
env.bak/
|
|
158
|
+
venv.bak/
|
|
159
|
+
|
|
160
|
+
# Spyder project settings
|
|
161
|
+
.spyderproject
|
|
162
|
+
.spyproject
|
|
163
|
+
|
|
164
|
+
# Rope project settings
|
|
165
|
+
.ropeproject
|
|
166
|
+
|
|
167
|
+
# mkdocs documentation
|
|
168
|
+
/site
|
|
169
|
+
|
|
170
|
+
# mypy
|
|
171
|
+
.mypy_cache/
|
|
172
|
+
.dmypy.json
|
|
173
|
+
dmypy.json
|
|
174
|
+
|
|
175
|
+
# Pyre type checker
|
|
176
|
+
.pyre/
|
|
177
|
+
|
|
178
|
+
# Ruff
|
|
179
|
+
.ruff_cache/
|
|
180
|
+
|
|
181
|
+
# pytype static type analyzer
|
|
182
|
+
.pytype/
|
|
183
|
+
|
|
184
|
+
# Cython debug symbols
|
|
185
|
+
cython_debug/
|
|
186
|
+
|
|
187
|
+
# PyCharm
|
|
188
|
+
# JetBrains specific template is maintained in a separate JetBrains.gitignore that can
|
|
189
|
+
# be found at https://github.com/github/gitignore/blob/main/Global/JetBrains.gitignore
|
|
190
|
+
# and can be added to the global gitignore or merged into this file. For a more nuclear
|
|
191
|
+
# option (not recommended) you can uncomment the following to ignore the entire idea folder.
|
|
192
|
+
.idea/
|
|
193
|
+
*.iml
|
|
194
|
+
|
|
195
|
+
# IDE Config
|
|
196
|
+
.vscode/
|
|
197
|
+
|
|
198
|
+
# Mac OS
|
|
199
|
+
.DS_Store
|
|
200
|
+
|
|
201
|
+
# Secrets
|
|
202
|
+
credentials.json
|
|
203
|
+
marin/crawl/bigquery-gcs-key.json
|
|
204
|
+
|
|
205
|
+
# Archive
|
|
206
|
+
archive/
|
|
207
|
+
|
|
208
|
+
# Caches and Outputs
|
|
209
|
+
!/scripts/web/output/
|
|
210
|
+
!/output/
|
|
211
|
+
|
|
212
|
+
# csv
|
|
213
|
+
*.csv
|
|
214
|
+
|
|
215
|
+
# wandb logs
|
|
216
|
+
wandb
|
|
217
|
+
artifacts
|
|
218
|
+
|
|
219
|
+
# Ignore generated credentials from google-github-actions/auth
|
|
220
|
+
gha-creds-*.json
|
|
221
|
+
|
|
222
|
+
.aider*
|
|
223
|
+
.git/*
|
|
224
|
+
|
|
225
|
+
*.jsonl
|
|
226
|
+
**/*.jsonl
|
|
227
|
+
# Echo's judged search corpus is source, not a run artifact.
|
|
228
|
+
!infra/marina/apps/echo/benchmarks/*.jsonl
|
|
229
|
+
scr/*
|
|
230
|
+
.weaver/
|
|
231
|
+
|
|
232
|
+
# Local host Marin config
|
|
233
|
+
.marin.yaml
|
|
234
|
+
|
|
235
|
+
/scratch
|
|
236
|
+
|
|
237
|
+
.forge
|
|
238
|
+
.claude/*
|
|
239
|
+
!.claude/skills
|
|
240
|
+
!.claude/agents
|
|
241
|
+
!.claude/settings.json
|
|
242
|
+
.agents/tmp/
|
|
243
|
+
.codex
|
|
244
|
+
.entire
|
|
245
|
+
.beads
|
|
246
|
+
|
|
247
|
+
.worktrees
|
|
248
|
+
.obsidian
|
|
249
|
+
.cw_env
|
|
250
|
+
.localbin/
|
|
251
|
+
|
|
252
|
+
# Stamped into built artifacts by lib/iris/hatch_build.py; never tracked.
|
|
253
|
+
lib/iris/src/iris/_build_info.py
|
|
@@ -0,0 +1,16 @@
|
|
|
1
|
+
Metadata-Version: 2.5
|
|
2
|
+
Name: marin-taskcompendium
|
|
3
|
+
Version: 0.2.160.dev38062449553
|
|
4
|
+
Summary: Semantic task contracts, submission extraction, and shared grading
|
|
5
|
+
Requires-Python: <3.14,>=3.12
|
|
6
|
+
Requires-Dist: marin-rigging==0.2.160.dev38062449553
|
|
7
|
+
Requires-Dist: marin-shellbox==0.2.160.dev38062449553
|
|
8
|
+
Requires-Dist: marin-verifyit[judge,math,reasoning-gym,schema]==0.2.160.dev38062449553
|
|
9
|
+
Requires-Dist: pydantic>=2.0
|
|
10
|
+
Requires-Dist: tomlkit>=0.13
|
|
11
|
+
Provides-Extra: pipeline
|
|
12
|
+
Requires-Dist: datasets<5.0.0,>=3.1.0; extra == 'pipeline'
|
|
13
|
+
Requires-Dist: marin-finestore==0.2.160.dev38062449553; extra == 'pipeline'
|
|
14
|
+
Requires-Dist: marin-zephyr==0.2.160.dev38062449553; extra == 'pipeline'
|
|
15
|
+
Requires-Dist: numpy>=1.26; extra == 'pipeline'
|
|
16
|
+
Requires-Dist: pyarrow>=15.0; extra == 'pipeline'
|
|
@@ -0,0 +1,356 @@
|
|
|
1
|
+
# TaskCompendium
|
|
2
|
+
|
|
3
|
+
The [task curation pipeline](../../docs/references/task-curation.md) downloads pinned sources, normalizes tasks, runs grading checks and GLM review, then writes final filtering decisions to sharded Parquet. Its audit retains every selected input, source locator, edit and rejection reason. Dataset declarations in the experiment name the pinned source, converter, rubric and grader controls; `taskcompendium.convert` holds the conversion techniques they share.
|
|
4
|
+
|
|
5
|
+
For ingestion work, start with the [pipeline overview](src/taskcompendium/pipeline/README.md)
|
|
6
|
+
and the [experiment flow](../../experiments/post_training/task_curation/README.md).
|
|
7
|
+
The sections below describe the task model and its presentation and grading contracts.
|
|
8
|
+
|
|
9
|
+
## What problem does it solve?
|
|
10
|
+
|
|
11
|
+
TaskCompendium stores a task's semantic contract, extracts one final submission with the task's answer format, and grades it with the task's grader. Expected answers, grader configuration, and verifier resources stay out of the model request.
|
|
12
|
+
|
|
13
|
+
Actor execution, model requests, tool dispatch, and actor environment lifecycle belong to the caller’s runtime. `taskcompendium.runtime` grades in a fresh Shellbox machine when the grader needs one. RolloutEngine provides the actor runtime; see [task rollouts](../../docs/references/task-rollouts.md). TaskCompendium does not depend on RolloutEngine.
|
|
14
|
+
|
|
15
|
+
## What does it contain?
|
|
16
|
+
|
|
17
|
+
- **Task specs** describe source problems, required capabilities, final results, answer formats, and graders.
|
|
18
|
+
- **Answer formats** describe how to request and extract a result, such as plain text, a JSON answer, or a final function call.
|
|
19
|
+
- **Typed evidence** records the conversation and an acquired text, JSON, action, or environment-state submission.
|
|
20
|
+
- **Grading** scores the extracted evidence in process or in a fresh grading machine, or reports that the task cannot be graded here.
|
|
21
|
+
|
|
22
|
+
```mermaid
|
|
23
|
+
flowchart LR
|
|
24
|
+
S[TaskSpec with answer format and grader] --> G[Extract once and grade]
|
|
25
|
+
A[Attempt evidence from runtime] --> G
|
|
26
|
+
G --> R[Typed grade result]
|
|
27
|
+
```
|
|
28
|
+
|
|
29
|
+
## What is a task spec?
|
|
30
|
+
|
|
31
|
+
`TaskSpec` defines one source task, including its grader and expected values. An importer or author creates it before choosing an execution runtime.
|
|
32
|
+
|
|
33
|
+
| Field | Meaning |
|
|
34
|
+
| --- | --- |
|
|
35
|
+
| `id` | Stable identity for this task. |
|
|
36
|
+
| `context` | The ordered model-visible conversation: text messages, historical assistant function calls, and tool results. |
|
|
37
|
+
| `environment_requirements` | Required capabilities, pinned initial workspace, and named tool-provider contracts. |
|
|
38
|
+
| `final_tools` | An ordered list of functions that terminate a chat. They are not backed by a tool provider. |
|
|
39
|
+
| `interaction_tools` | Executable function declarations used by the optional episode runtime. |
|
|
40
|
+
| `output_paths` | Absolute output paths captured by the optional episode runtime, outside `/tests` and `/logs/verifier`. |
|
|
41
|
+
| `output_directories` | Workspace roots, relative fnmatch patterns and explicit file-count/aggregate-byte capture budgets. |
|
|
42
|
+
| `answer_type` | The semantic result: `text`, `number`, `json`, `file`, `state`, `workspace_state`, or `native_action`. |
|
|
43
|
+
| `answer_format` | How the final answer is requested from the model and extracted. See [Answer formats](#answer-formats). |
|
|
44
|
+
| `grader` | How an attempt is graded. See [What is a grader?](#what-is-a-grader) |
|
|
45
|
+
| `source` | Upstream dataset, revision, row, and importer revision retained as audit provenance. |
|
|
46
|
+
| `schema_version` | Version of the serialized spec: `0.25`. Readers reject other versions. |
|
|
47
|
+
| `resources` | Inline files grouped under `all`, `worker`, `oracle`, and `verifier` visibility. |
|
|
48
|
+
| `tags` | Arbitrary descriptive strings, retained in order, including duplicates and empty strings. |
|
|
49
|
+
|
|
50
|
+
A task has one final result. TaskSpec does not define ordered task stages or stage-reward aggregation.
|
|
51
|
+
|
|
52
|
+
Directory capture requires the actor's `python3` capability and a real POSIX
|
|
53
|
+
Python interpreter; ShellSim does not support it. Selection patterns use
|
|
54
|
+
case-sensitive fnmatch semantics, where `*` includes `/` and hidden paths.
|
|
55
|
+
Capture preserves unsorted depth-first filesystem order, skips symlinks in
|
|
56
|
+
directory selections, and fails the whole capture on file or byte overflow.
|
|
57
|
+
Existing named `output_paths` retain their file/symlink behavior. Roots must stay
|
|
58
|
+
inside the grader's workspace and outside `/tests`, `/logs/verifier` and
|
|
59
|
+
`/solution`. Membership and budgets are checked again before files are staged for grading.
|
|
60
|
+
|
|
61
|
+
`context.events` is the model-visible conversation prefix. A text event retains its role and content. Historical assistant calls and tool results retain their call IDs and order; a runtime preserves this history when presenting the task to the model. `answer_type` does not prescribe a wrapper such as JSON; `answer_format` does.
|
|
62
|
+
|
|
63
|
+
`ConversationInput` is not a lossless Responses API transcript. It excludes provider reasoning items. The NeMo importer omits an unencrypted historical reasoning summary only when a visible assistant message or function call follows it. It rejects encrypted reasoning and reasoning left at the decision point. Exact provider continuation from reasoning state is outside this contract.
|
|
64
|
+
|
|
65
|
+
An **answer** is the task’s semantic result, identified by `answer_type`. A **submission** is the typed value the task's answer format extracts from a completed attempt for grading. Different answer formats can yield the same submission and carry the same answer.
|
|
66
|
+
|
|
67
|
+
For example, a task asking “What is 7 + 5?” can have `answer_type=number` and a `numeric` grader that expects `12`. The `PlainText`, `JsonAnswer`, and `AnswerCall` formats carry that answer as `12`, `{"answer":"12"}`, and a final `submit_answer({"answer":"12"})` call, respectively. Each extracts the string `12` for the same grader, which accepts both `12` and `12.0`. A task asking for a function call has `answer_type=native_action`; its context retains the conversation and its `final_tools` field describes the source functions. The grader and expected answer are never added to the model-visible input. Importers must make source output instructions agree with the task's answer format, or reject rows they cannot safely rewrite. A raw-output requirement in a source message would conflict with `JsonAnswer` or `AnswerCall`; `answer_type=text` alone cannot detect that conflict in prose.
|
|
68
|
+
|
|
69
|
+
## What can a task represent?
|
|
70
|
+
|
|
71
|
+
### Text answers
|
|
72
|
+
|
|
73
|
+
A text task uses `answer_type=text`. `PlainText`, `Boxed`, `JsonAnswer`, and `AnswerCall` can carry its answer. The `exact` mode compares normalized text; `mcq` grades a single option letter. Invalid MCQ option text scores zero through verifyit. Both modes grade the answer extracted by any of these formats.
|
|
74
|
+
|
|
75
|
+
### Numeric answers
|
|
76
|
+
|
|
77
|
+
A numeric task uses `answer_type=number` and can use the same answer formats as text. Expected values are required numeric literal strings, preserving integers, decimals and fractions exactly. Absolute and relative tolerances are required finite nonnegative floats. The `numeric` mode reads the last boxed answer, or exactly one numeric literal from the last nonempty line when no box is present. Surrounding prose, including negation, is ignored. Missing, malformed, or ambiguous numeric output is `submission_failure`; a valid wrong number is `graded` with reward `0.0`.
|
|
78
|
+
|
|
79
|
+
`taskcompendium.grading.grade_result` maps verifyit rewards to `GradeResult` the same way for in-process grading and for the verdict the verifyit command writes in a grading machine. Malformed numeric text is `submission_failure` with reward `0.0` in both.
|
|
80
|
+
|
|
81
|
+
### Final function calls
|
|
82
|
+
|
|
83
|
+
A task whose result is a function call uses `answer_type=native_action` and the `FinalAction` answer format. Its `final_tools` field declares the available functions. `FinalAction` defines whether a call is required and the maximum call count. It extracts the assistant's final message, and the `predicted_action` mode compares the submitted function names and decoded argument objects with the expected calls. Grading compares the submitted calls without dispatching them.
|
|
84
|
+
|
|
85
|
+
For text and numeric tasks, the `AnswerCall` format adds `submit_answer(answer: string)` alongside the task’s final tools. It extracts the call's `answer` argument and passes it to the task's grader. A non-call response or a call to another function receives `submission_failure` with reward `0.0`.
|
|
86
|
+
|
|
87
|
+
### Files and state
|
|
88
|
+
|
|
89
|
+
`answer_type=file` names a file result. `answer_type=workspace_state` names the final filesystem workspace. `answer_type=state` names arbitrary resulting environment state, including provider state outside a filesystem. Acquiring these results requires an actor runtime. These tasks still declare an `answer_format`, but grading does not read it. `file` and `workspace_state` answers require a grader with an environment. In-process grading uses supplied evidence; the optional TaskCompendium episode runtime captures declared output files. The `structured_exact` mode compares JSON values and ordered arrays. Numbers compare by value by default (`16` equals `16.0`); booleans remain distinct. Set `numeric_types="strict"` to require exact numeric scalar types.
|
|
90
|
+
|
|
91
|
+
`answer_type=json` is a JSON answer from the model, independent of environment state. `JsonValueAnswer` parses the complete final chat text into a `JsonSubmission` for `structured_exact`. `StateSubmission` is evidence acquired from an environment by its runtime. `structured_exact` accepts either. The `JsonAnswer` format instead unwraps an `answer` string for text or numeric tasks. Both formats reject duplicate keys at every nesting level and nonfinite numbers. Tool-call arguments are decoded before entering the conversation evidence. Runtimes own provider-response decoding and may report malformed tool-call arguments as a structural submission failure with reward `0.0`. Both historical and final calls contain typed argument objects.
|
|
92
|
+
|
|
93
|
+
Public expectations belong in `context`: for example, the columns a CSV must contain or the behavior a repaired project must provide. The grader checks those expectations. The answer format chooses how the result is delivered and extracted. `answer_type` identifies the answer’s semantic kind.
|
|
94
|
+
|
|
95
|
+
## Environment requirements
|
|
96
|
+
|
|
97
|
+
`environment_requirements` declares the initial state and operations needed to solve a task.
|
|
98
|
+
|
|
99
|
+
`compatible_backends` lists the Shellbox backends the source author permits for
|
|
100
|
+
this environment, such as `shellsim` or `gvisor`. A grader environment has its own
|
|
101
|
+
list. Empty means no Shellbox backend is declared; it is not a wildcard.
|
|
102
|
+
TaskCompendium's shell runtime and `grade_task` check the selected factory against
|
|
103
|
+
the appropriate list before creating a machine. A required Docker image excludes ShellSim. See the
|
|
104
|
+
[source-author rubric](src/taskcompendium/pipeline/README.md) for compatibility
|
|
105
|
+
criteria and the distinction between declarations and sampled runtime evidence.
|
|
106
|
+
|
|
107
|
+
| Field | Meaning |
|
|
108
|
+
| --- | --- |
|
|
109
|
+
| `capabilities` | Unique operation names, such as `shell`, `network`, `filesystem`, `process`, or `browser`. Names are open so future capabilities can be represented. |
|
|
110
|
+
| `compatible_backends` | Unique Shellbox backend names permitted by the source author; empty declares none. |
|
|
111
|
+
| `docker_image` | Optional immutable image reference, such as `registry/project@sha256:<64 lowercase hex digits>`. Tags alone are rejected. |
|
|
112
|
+
| `docker_build` | Optional `DockerBuildContext(files=...)` containing inline regular files relative to the build-context root, including `Dockerfile`. The recipe is unresolved data. |
|
|
113
|
+
| `working_directory` | Optional normalized absolute POSIX path for the main workspace. Omission declares no required working directory. |
|
|
114
|
+
| `setup_commands` | Ordered commands required to establish the initial workspace. |
|
|
115
|
+
| `environment_variables` | String values required in the worker or grading machine environment. |
|
|
116
|
+
| `tool_providers` | Mapping from a task-local provider instance name to a required action interface and initial state. |
|
|
117
|
+
| `packages_lock` | Storage URL of a uv-compiled, hash-pinned requirements lock. A `local` environment requires it: the host builds the lock into a Python environment for the commands it runs. Other environments must omit it. |
|
|
118
|
+
|
|
119
|
+
Each `ProviderRequirement` contains `action_interface`, a versioned contract name such as `workplace:v1`, and required `initial_state`, a JSON value such as a string, null, or an object. Two named instances can require the same interface with different initial states. No digest is required. The selected runtime owns provider implementation, transport, state initialization, reset, and tool execution. `final_tools` contains only ordered function definitions advertised at the final decision point; it supplies no implementation.
|
|
120
|
+
|
|
121
|
+
The task's environment and worker file mounts describe worker initial state. A grader that runs in its own machine declares that machine's image or build recipe, capabilities, setup commands, and environment variables in `grader.environment`. A grader environment requires a `docker_image`, `docker_build`, or local `packages_lock`, and cannot declare tool providers. Runtimes keep those requirements and verifier resources separate from the worker environment.
|
|
122
|
+
|
|
123
|
+
`docker_image`, `docker_build`, and `packages_lock` are mutually exclusive.
|
|
124
|
+
Build files preserve bytes, modes and timestamps; paths cannot collide. Resource
|
|
125
|
+
budgets count actor and grader contexts separately. Execution requires the caller
|
|
126
|
+
to build each context and replace it with a digest-pinned image and supported
|
|
127
|
+
backends. TaskCompendium supplies no build resolver; runtimes and SAMPLE/FULL
|
|
128
|
+
controls reject unresolved contexts. Local and ShellSim backends cannot declare them.
|
|
129
|
+
|
|
130
|
+
The [Harbor exporter](src/taskcompendium/harbor/export.py) emits build inputs;
|
|
131
|
+
see the [campaign quickstart](../../experiments/post_training/task_curation/README.md#harbor-compatibility-view)
|
|
132
|
+
for execution limits. The Harbor-to-TaskSpec importer requires a prebuilt image.
|
|
133
|
+
|
|
134
|
+
## Resource mounts
|
|
135
|
+
|
|
136
|
+
`resources` is a `ResourceGroups` object. Each group contains an ordered list of `TaskResource` mounts.
|
|
137
|
+
|
|
138
|
+
| Group | Visibility |
|
|
139
|
+
| --- | --- |
|
|
140
|
+
| `all` | Shared inputs visible to the worker, oracle, and grader. |
|
|
141
|
+
| `worker` | Model-visible inputs. |
|
|
142
|
+
| `oracle` | Reference material for oracle controls, hidden from the model. |
|
|
143
|
+
| `verifier` | Grader inputs, hidden from the model. |
|
|
144
|
+
|
|
145
|
+
Gold answers and hidden tests belong in `oracle` or `verifier`. An `all` resource is model-visible. A role receives `all` followed by its own mounts. Destinations must be distinct in that combined sequence and have no file/directory ancestor collisions. Names retain POSIX case distinctions. Separate role-specific groups can reuse a relative path without sharing their content.
|
|
146
|
+
|
|
147
|
+
Oracle resources are reserved for trusted reference-solution generation. Verifier resources are used when evaluating a candidate result. These groups declare access; they do not require an oracle or evaluator process to run.
|
|
148
|
+
|
|
149
|
+
The TaskCompendium and RolloutEngine runtimes install shared, worker, and oracle resources at `/<path>`: a resource at `app/project/input.txt` appears at `/app/project/input.txt`. Worker mounts are established before setup commands run. A grading machine receives verifier resources at `/tests/<path>`; in-process modes read them by that relative path.
|
|
150
|
+
|
|
151
|
+
Each `TaskResource` contains one inline file, with these fields:
|
|
152
|
+
|
|
153
|
+
| Field | Meaning |
|
|
154
|
+
| --- | --- |
|
|
155
|
+
| `path` | Normalized relative destination: `/<path>` for shared, worker, and oracle resources, and `/tests/<path>` for verifier resources. |
|
|
156
|
+
| `source` | An `InlineFile` containing the exact file bytes, encoded as canonical base64. |
|
|
157
|
+
| `mode` | Optional Unix permission mode as a three- or four-digit octal string, such as `0644` or `0755`. An explicit mode applies to the mounted file. Files default to `0644` when the mode is omitted. |
|
|
158
|
+
| `mtime_ns` | Optional integer Unix modification timestamp in nanoseconds for the mounted file. Omission leaves the timestamp unspecified. |
|
|
159
|
+
|
|
160
|
+
`InlineFile` has `kind="inline_file"` and `content_base64`. UTF-8 text uses the same byte representation as binary files. Resource contents are stored in the task spec; no dataset root or process working directory is used to locate them. Shared external files are deferred until a `TaskSet` contract defines their location and loading.
|
|
161
|
+
|
|
162
|
+
An archive can be included as ordinary file bytes, but resource mounting does not extract it. Paths may contain directories to locate the file within the workspace. Directory resources and recursive copies are unsupported. Schema decoding performs no filesystem inspection or I/O.
|
|
163
|
+
|
|
164
|
+
Grouped inline resources can contain:
|
|
165
|
+
|
|
166
|
+
```json
|
|
167
|
+
{
|
|
168
|
+
"resources": {
|
|
169
|
+
"all": [
|
|
170
|
+
{
|
|
171
|
+
"path": "README.txt",
|
|
172
|
+
"source": {"kind": "inline_file", "content_base64": "VXNlIHRoZSBzdXBwbGllZCBwcm9qZWN0Lg=="},
|
|
173
|
+
"mode": "0444",
|
|
174
|
+
"mtime_ns": 1725555600000000000
|
|
175
|
+
}
|
|
176
|
+
],
|
|
177
|
+
"worker": [
|
|
178
|
+
{"path": "project/input.txt", "source": {"kind": "inline_file", "content_base64": "cHVibGljIGlucHV0"}}
|
|
179
|
+
],
|
|
180
|
+
"oracle": [
|
|
181
|
+
{"path": "answer.txt", "source": {"kind": "inline_file", "content_base64": "cHJpdmF0ZSByZWZlcmVuY2U="}}
|
|
182
|
+
],
|
|
183
|
+
"verifier": [
|
|
184
|
+
{"path": "checks/grade.py", "source": {"kind": "inline_file", "content_base64": "cHJpdmF0ZSBjaGVja3M="}}
|
|
185
|
+
]
|
|
186
|
+
}
|
|
187
|
+
}
|
|
188
|
+
```
|
|
189
|
+
|
|
190
|
+
File materializers must reject unsafe destinations and collisions, and enforce byte limits. The schema and in-process grading do not materialize resources. Execution runtimes supply mounts and role isolation; `grade_in_sandbox` stages grading inputs into a fresh machine.
|
|
191
|
+
|
|
192
|
+
## What can we import?
|
|
193
|
+
|
|
194
|
+
### TaskTrove MCQA
|
|
195
|
+
|
|
196
|
+
The TaskTrove MCQA importer reads archives from a cleaned release. See the [published TaskTrove Clean dataset](https://huggingface.co/datasets/open-athena/task-trove). Its caller passes the archive bytes, upstream subset, archive path, and release provenance to `read_archive`. The reader checks the subset and path against the archive manifest; the release URI and revision are caller-supplied provenance. The importer checks the source answer-line template before replacing it with a one-letter instruction. Its text answer uses the `PlainText` format. The `mcq` grader stores the expected letter and option count and grades in process with the `verifyit` MCQ scorer. This importer supports only MCQ mode. Executable TaskTrove modes need verifier resources and a grading machine.
|
|
197
|
+
|
|
198
|
+
### NeMo predicted function calls
|
|
199
|
+
|
|
200
|
+
`taskcompendium.importers.nemo_predicted_action.import_row` accepts a NeMo predicted-function-call row and a caller-pinned digest of that row. `canonical_sha256(row)` hashes its UTF-8 JSON with sorted keys and compact separators; record the digest with the source revision before importing. The importer returns a TaskSpec with `answer_type=native_action` and a `FinalAction` answer format. The context carries the source conversation; `final_tools` carries advertised functions. `FinalAction.require_call` is set when the source requires a tool call, and `FinalAction.max_calls=1` when the source disables parallel calls. The expected function calls stay in the `predicted_action` grader. There is one stored conversation, with no second flattened prompt to keep in sync.
|
|
201
|
+
|
|
202
|
+
The runtime presents the source turns and function definitions and retains the final response as typed evidence. The grader compares submitted function names and JSON arguments. The importer rejects rows whose expected action is an assistant text message because the source comparator gives any message full credit; it also rejects request settings it cannot carry. The pinned fixture records the NeMo Gym repository revision and blob SHA in `tests/fixtures/nemo/predicted-action.provenance.json`. Numeric tolerance is used only when explicitly set in the grader's parameters.
|
|
203
|
+
|
|
204
|
+
## What is a grader?
|
|
205
|
+
|
|
206
|
+
`TaskSpec.grader` declares how an attempt is graded. Its `kind` field selects one of four classes in `taskcompendium.models`:
|
|
207
|
+
|
|
208
|
+
| `kind` | Class | Grading |
|
|
209
|
+
| --- | --- | --- |
|
|
210
|
+
| `verifyit` | `VerifyitGrader` | A stock verifyit `mode` and its `parameters`: the keys of verifyit's `verifier.toml` table other than `mode`. |
|
|
211
|
+
| `script` | `ScriptGrader` | A command run after the episode in a fresh machine built from a pinned image. |
|
|
212
|
+
| `session` | `SessionGrader` | A registered custom RolloutEngine `TaskSession` grades the task. |
|
|
213
|
+
| `none` | `NoGrader` | The task cannot be graded here. `reason` says why; `contract` optionally keeps the source's grading terms. |
|
|
214
|
+
|
|
215
|
+
A `VerifyitGrader` without an `environment` grades the extracted answer in process. Its mode must be one of `verifyit.candidate.IN_PROCESS_MODES`: `exact`, `numeric`, `mcq`, `math`, `ifeval`, `json-schema`, `xml-elements`, `csv-columns`, `structured_exact`, or `predicted_action`. A `VerifyitGrader` with an `environment` runs the verifyit command in a fresh machine built from that environment, with verifier resources under `/tests` and the verifyit specification at `/tests/verifier.toml`. Modes that execute code or call a model, such as `pytest`, `stdio`, and `judge`, grade only there. Construction validates `parameters` with verifyit.
|
|
216
|
+
|
|
217
|
+
A `ScriptGrader` declares:
|
|
218
|
+
|
|
219
|
+
| Field | Meaning |
|
|
220
|
+
| --- | --- |
|
|
221
|
+
| `argv`, `cwd`, `env` | The grader command, its working directory (default `/app`), and added environment variables. |
|
|
222
|
+
| `environment` | The grading machine's requirements. |
|
|
223
|
+
| `collect` | Commands run as root on the agent's machine before grading. |
|
|
224
|
+
| `artifacts` | Files and directories copied from the agent's machine into the grading machine, with exclusions and a missing-file policy. |
|
|
225
|
+
| `answer_path` | Where the extracted answer is written (default `/app/answer.txt`): text as text, JSON as JSON, and a final action as the final assistant message's JSON. |
|
|
226
|
+
| `conversation_path` | Where the conversation is written as OpenAI-style chat messages in JSON (default `/tests/conversation.json`). |
|
|
227
|
+
| `reward` | `StdoutReward` (default), `ExitCodeReward`, or `FileReward`. |
|
|
228
|
+
| `timeout` | Deadline in seconds for each grading command (default 600). |
|
|
229
|
+
|
|
230
|
+
`StdoutReward` requires exit code 0 and one finite number on standard output; anything else is `infra_error`. `ExitCodeReward` scores 1.0 for exit code 0 and 0.0 otherwise. `FileReward.files` lists candidate reward files in priority order; the first that exists supplies the score. Each `RewardFile` holds a number, or a JSON object whose `key` (default `reward`) holds the score and whose optional `detail` object becomes the grade detail. `FileReward.pass_above` sets `GradeResult.passed`; `ExitCodeReward` sets it from the exit code. A missing, empty, or invalid reward file is `infra_error`. A timed-out command is `infra_error` for every reward source.
|
|
231
|
+
|
|
232
|
+
TaskSpec validation ties the grader to the answer:
|
|
233
|
+
|
|
234
|
+
- An in-process `VerifyitGrader` cannot grade `file` or `workspace_state` answers.
|
|
235
|
+
- A `VerifyitGrader` with an environment cannot grade `native_action` answers. Its workspace modes (`stdio`, `pytest`, `junit`, and `gotest`) cannot grade `text`, `number`, or `json` answers.
|
|
236
|
+
- `ScriptGrader.answer_path` must be `None` for `file`, `state`, and `workspace_state` answers. `conversation_path` must not replace a verifier resource.
|
|
237
|
+
- Answer paths, output paths, and output directories must lie outside `/tests` and `/logs/verifier`.
|
|
238
|
+
|
|
239
|
+
A `SessionGrader` task is graded by the registered custom `TaskSession` that runs it; TaskCompendium cannot grade it. A `NoGrader` task grades as `unavailable`, with no reward and its `reason` as the error. Converters use `NoGrader` when the source evaluator needs a runtime this repository cannot provide; its `contract` records the evaluator, source revision, grading data and runtime requirements. `taskcompendium.grader.grader_config(task)` returns a copy of a `NoGrader`'s contract, or the `config.json` verifier resource of another grader.
|
|
240
|
+
|
|
241
|
+
`taskcompendium.grader.GraderPackage` pairs a grader with its verifier resources, whose paths are relative to `/tests`. `verifyit_package(spec, resources=(), environment=None)` builds a `VerifyitGrader` from a verifyit `Spec`. Build other packages directly with `GraderPackage(grader, resources)`.
|
|
242
|
+
|
|
243
|
+
```python
|
|
244
|
+
from taskcompendium.grader import verifyit_package
|
|
245
|
+
from verifyit.spec import ExactSpec, FunctionCall, McqSpec, PredictedActionSpec, StructuredExactSpec
|
|
246
|
+
|
|
247
|
+
text_grader = verifyit_package(ExactSpec(expected=("expected text",))).grader
|
|
248
|
+
mcq_grader = verifyit_package(McqSpec(expected="C", options=4)).grader
|
|
249
|
+
json_grader = verifyit_package(StructuredExactSpec(expected={"value": 16})).grader
|
|
250
|
+
action_grader = verifyit_package(
|
|
251
|
+
PredictedActionSpec(expected_calls=(FunctionCall(name="lookup", arguments={"city": "Paris"}),))
|
|
252
|
+
).grader
|
|
253
|
+
```
|
|
254
|
+
|
|
255
|
+
A grader scores one acquired answer. Comparative scoring across several attempts, cohort membership, and grading phase belong to the trainer. Standard grading modes and in-process scoring live in `verifyit`; TaskCompendium has no grader registry or separate standard verifier schema.
|
|
256
|
+
|
|
257
|
+
## Answer formats
|
|
258
|
+
|
|
259
|
+
`TaskSpec.answer_format` says how the final answer is requested from the model and extracted from `conversation.events[-1]`, the final assistant message or call batch. Earlier events are validated history; historical calls must have their results before the final submission.
|
|
260
|
+
|
|
261
|
+
| Format | `kind` | Extracts | Answer types |
|
|
262
|
+
| --- | --- | --- | --- |
|
|
263
|
+
| `PlainText` | `plain_text` | The whole final message | `text`, `number` |
|
|
264
|
+
| `Boxed` | `boxed` | The content of the last `\boxed{...}`, else the whole final message | `text`, `number` |
|
|
265
|
+
| `JsonAnswer` | `json_answer` | The nonempty string `answer` field of a JSON object | `text`, `number` |
|
|
266
|
+
| `JsonValueAnswer` | `json_value` | The whole final message as one JSON value | `json` |
|
|
267
|
+
| `AnswerCall` | `answer_call` | The `answer` argument of one `submit_answer` call | `text`, `number` |
|
|
268
|
+
| `FinalAction` | `final_action` | The final message: calls to `final_tools`, or text | `native_action` |
|
|
269
|
+
|
|
270
|
+
A `text`, `number`, `json`, or `native_action` task must use a format that carries its answer type. `submission_compatibility(task)` also checks that the grader's verifyit mode accepts the submission the format extracts (`TextSubmission`, `JsonSubmission`, or `ActionSubmission`), that a `FinalAction` task has final tools, and that no final tool is named `submit_answer` under `AnswerCall`.
|
|
271
|
+
|
|
272
|
+
Answer formats preserve the task's advertised functions. `AnswerCall` adds `submit_answer` with an answer string. `FinalAction(require_call=True)` requires a call, and `FinalAction(max_calls=1)` limits the submission to one call. Missing required calls and excessive call counts are `submission_failure` with reward `0.0`.
|
|
273
|
+
|
|
274
|
+
`taskcompendium.submission` builds model requests from the task alone. `submission_instruction(answer_format)` returns the format's final-answer instruction, and `answer_call_tool()` returns the `submit_answer` tool definition. `chat_request(task)` and `render_instruction(task)` combine them with the public conversation and final tools. These helpers perform no execution.
|
|
275
|
+
|
|
276
|
+
## Grading
|
|
277
|
+
|
|
278
|
+
Create a task:
|
|
279
|
+
|
|
280
|
+
```python
|
|
281
|
+
from taskcompendium.grader import verifyit_package
|
|
282
|
+
from taskcompendium.models import (
|
|
283
|
+
AnswerType,
|
|
284
|
+
ConversationInput,
|
|
285
|
+
EnvironmentRequirements,
|
|
286
|
+
PlainText,
|
|
287
|
+
Source,
|
|
288
|
+
TaskSpec,
|
|
289
|
+
TextMessage,
|
|
290
|
+
)
|
|
291
|
+
from verifyit.spec import NumericSpec
|
|
292
|
+
|
|
293
|
+
spec = TaskSpec(
|
|
294
|
+
id="arithmetic-7-plus-5",
|
|
295
|
+
context=ConversationInput(events=(TextMessage(role="user", content="What is 7 + 5?"),)),
|
|
296
|
+
environment_requirements=EnvironmentRequirements(),
|
|
297
|
+
answer_type=AnswerType.NUMBER,
|
|
298
|
+
answer_format=PlainText(),
|
|
299
|
+
grader=verifyit_package(NumericSpec(expected="12", tolerance_abs=0.0, tolerance_rel=0.0)).grader,
|
|
300
|
+
source=Source(dataset="hand-authored", revision="2026-09-16", row="arithmetic-7-plus-5", importer_revision="1"),
|
|
301
|
+
)
|
|
302
|
+
```
|
|
303
|
+
|
|
304
|
+
After execution, the caller supplies the complete conversation to `GradingAttempt`:
|
|
305
|
+
|
|
306
|
+
```python
|
|
307
|
+
from taskcompendium.grading import grade_answer
|
|
308
|
+
from taskcompendium.models import ConversationTrace, GradingAttempt
|
|
309
|
+
|
|
310
|
+
def score_final_response(content: str):
|
|
311
|
+
conversation = ConversationTrace(
|
|
312
|
+
events=(*spec.context.events, TextMessage(role="assistant", content=content))
|
|
313
|
+
)
|
|
314
|
+
return grade_answer(spec, GradingAttempt(conversation))
|
|
315
|
+
```
|
|
316
|
+
|
|
317
|
+
Three functions grade an attempt:
|
|
318
|
+
|
|
319
|
+
- `taskcompendium.grading.grade_answer(task, attempt)` grades with an in-process `VerifyitGrader`. The answer format extracts the submission, or `attempt.state` supplies it for a `state` answer, and `verifyit.candidate.grade_candidate` scores it. It raises `TypeError` for other graders.
|
|
320
|
+
- `taskcompendium.runtime.grading.grade_in_sandbox(task, attempt, factory, machine_spec, *, task_machine=None, timeout=None)` is asynchronous. It grades with a `VerifyitGrader` that has an environment, or with a `ScriptGrader`, in a fresh machine from `factory`. `task_machine` is the agent's machine, required for `collect` and `artifacts`. Machine exceptions propagate.
|
|
321
|
+
- `taskcompendium.runtime.task_grading.grade_task(task, attempt, *, machine_factory=None, machine_spec=None)` grades any task synchronously. A `NoGrader` task is `unavailable`. An in-process grader uses `grade_answer`. A grader with an environment uses `grade_in_sandbox` after `grade_task` checks the factory's backend against `environment.compatible_backends`. A missing factory or machine specification, and a machine `RuntimeError` or `OSError`, become `infra_error`. A `SessionGrader` raises `TypeError`.
|
|
322
|
+
|
|
323
|
+
RolloutEngine's Shellbox session calls `grade_answer` or `grade_in_sandbox` after the turn loop; see [task rollouts](../../docs/references/task-rollouts.md).
|
|
324
|
+
|
|
325
|
+
The execution runtime completes file and environment-state acquisition before grading. `GradingAttempt` carries the conversation, captured file bytes keyed by absolute path, and an optional `StateSubmission`. A missing state is `None`; captured JSON null is `StateSubmission(None)`. `taskcompendium.runtime.models.grading_attempt(conversation, evidence)` builds an attempt from captured `RuntimeEvidence`. A grading machine also receives a conversational answer extracted from the conversation. Expected values remain in `task.grader` and verifier resources and must not be included in the model request.
|
|
326
|
+
|
|
327
|
+
A valid correct answer produces `GradeResult(status=graded, reward=1.0)`; a valid wrong answer produces `graded` with reward `0.0`. A final message the answer format cannot read, and numeric text the `numeric` mode cannot read, produce `submission_failure` with reward `0.0`. verifyit's `invalid_task` and `infra_error` statuses carry no reward and stay separate from wrong or invalid submissions. `grade_answer` raises when the answer format is incompatible with the grader.
|
|
328
|
+
|
|
329
|
+
Execution runtimes decode provider responses into `TextMessage` or `AssistantToolCalls` before constructing a `ConversationTrace`. Tool-call arguments must be decoded JSON objects. A runtime may expose malformed provider output as `submission_failure` with reward `0.0`; network and execution errors remain separate.
|
|
330
|
+
|
|
331
|
+
## Dataset conversion
|
|
332
|
+
|
|
333
|
+
TaskSpec defines the serialized task contract. Dataset conversion pipelines own storage layout and streaming I/O, using Zephyr for Parquet processing. The optional `taskcompendium.pipeline` package stores curation outputs through Zephyr. See [pipeline contracts](src/taskcompendium/pipeline/README.md). JSON decoding preserves valid requirements independently of a runtime's support for them.
|
|
334
|
+
|
|
335
|
+
```python
|
|
336
|
+
serialized = spec.model_dump_json()
|
|
337
|
+
restored = TaskSpec.model_validate_json(serialized)
|
|
338
|
+
```
|
|
339
|
+
|
|
340
|
+
The serialized spec includes the grader configuration and verifier resources. Store it where trusted grading code can read it; build model requests with `chat_request(task)` or `render_instruction(task)`, which read only the public context, final tools, and answer format.
|
|
341
|
+
|
|
342
|
+
## Development
|
|
343
|
+
|
|
344
|
+
TaskCompendium requires Python 3.12 or 3.13 and uses `marin-rigging` for relative POSIX path and mount-collision validation. Resource names preserve Linux semantics, including colons, backslashes, trailing spaces, and case distinctions. Absolute paths, traversal, NUL bytes, duplicate files, and file/directory collisions are rejected. The validator leaf module performs no storage access.
|
|
345
|
+
|
|
346
|
+
TaskCompendium uses the root workspace's `uv.lock` and `.venv`. Run the package tests from the repository root:
|
|
347
|
+
|
|
348
|
+
```bash
|
|
349
|
+
uv run --package marin-taskcompendium --extra pipeline --group test pytest lib/taskcompendium/tests -q
|
|
350
|
+
|
|
351
|
+
# Type-check from the package project directory with its dependencies available.
|
|
352
|
+
cd lib/taskcompendium
|
|
353
|
+
uvx --from 'pyrefly>=1.0.0,<1.1.0' pyrefly check
|
|
354
|
+
```
|
|
355
|
+
|
|
356
|
+
Schema `0.25` stores the answer format and the typed grader on each task. Decoders reject other schema versions; existing conversion pipelines must emit the current contract.
|