marin-taskcompendium 0.2.160.dev38062449553__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (119) hide show
  1. marin_taskcompendium-0.2.160.dev38062449553/.gitignore +253 -0
  2. marin_taskcompendium-0.2.160.dev38062449553/PKG-INFO +16 -0
  3. marin_taskcompendium-0.2.160.dev38062449553/README.md +356 -0
  4. marin_taskcompendium-0.2.160.dev38062449553/pyproject.toml +58 -0
  5. marin_taskcompendium-0.2.160.dev38062449553/src/taskcompendium/__init__.py +2 -0
  6. marin_taskcompendium-0.2.160.dev38062449553/src/taskcompendium/chat.py +93 -0
  7. marin_taskcompendium-0.2.160.dev38062449553/src/taskcompendium/convert/__init__.py +4 -0
  8. marin_taskcompendium-0.2.160.dev38062449553/src/taskcompendium/convert/answers.py +210 -0
  9. marin_taskcompendium-0.2.160.dev38062449553/src/taskcompendium/convert/code.py +78 -0
  10. marin_taskcompendium-0.2.160.dev38062449553/src/taskcompendium/convert/conversation.py +44 -0
  11. marin_taskcompendium-0.2.160.dev38062449553/src/taskcompendium/convert/delivery.py +34 -0
  12. marin_taskcompendium-0.2.160.dev38062449553/src/taskcompendium/convert/environment.py +16 -0
  13. marin_taskcompendium-0.2.160.dev38062449553/src/taskcompendium/convert/executable.py +348 -0
  14. marin_taskcompendium-0.2.160.dev38062449553/src/taskcompendium/convert/json_schema.py +32 -0
  15. marin_taskcompendium-0.2.160.dev38062449553/src/taskcompendium/convert/nemotron_ultra.py +225 -0
  16. marin_taskcompendium-0.2.160.dev38062449553/src/taskcompendium/convert/preference.py +87 -0
  17. marin_taskcompendium-0.2.160.dev38062449553/src/taskcompendium/convert/script_grader.py +61 -0
  18. marin_taskcompendium-0.2.160.dev38062449553/src/taskcompendium/convert/tasktrove.py +184 -0
  19. marin_taskcompendium-0.2.160.dev38062449553/src/taskcompendium/convert/tasktrove_code_contests.py +99 -0
  20. marin_taskcompendium-0.2.160.dev38062449553/src/taskcompendium/convert/tasktrove_codeforces.py +117 -0
  21. marin_taskcompendium-0.2.160.dev38062449553/src/taskcompendium/convert/tasktrove_converted_task.py +88 -0
  22. marin_taskcompendium-0.2.160.dev38062449553/src/taskcompendium/convert/tasktrove_json_schemas.py +133 -0
  23. marin_taskcompendium-0.2.160.dev38062449553/src/taskcompendium/convert/tasktrove_nemotron_data.py +25 -0
  24. marin_taskcompendium-0.2.160.dev38062449553/src/taskcompendium/convert/tasktrove_nemotron_structured_outputs.py +284 -0
  25. marin_taskcompendium-0.2.160.dev38062449553/src/taskcompendium/convert/tasktrove_nl2bash.py +168 -0
  26. marin_taskcompendium-0.2.160.dev38062449553/src/taskcompendium/convert/tasktrove_python_unit_tests.py +100 -0
  27. marin_taskcompendium-0.2.160.dev38062449553/src/taskcompendium/convert/tasktrove_stdio_cases.py +91 -0
  28. marin_taskcompendium-0.2.160.dev38062449553/src/taskcompendium/convert/tasktrove_taco.py +88 -0
  29. marin_taskcompendium-0.2.160.dev38062449553/src/taskcompendium/convert/verifyit_build.py +62 -0
  30. marin_taskcompendium-0.2.160.dev38062449553/src/taskcompendium/direct_chat.py +40 -0
  31. marin_taskcompendium-0.2.160.dev38062449553/src/taskcompendium/grader.py +42 -0
  32. marin_taskcompendium-0.2.160.dev38062449553/src/taskcompendium/grading.py +118 -0
  33. marin_taskcompendium-0.2.160.dev38062449553/src/taskcompendium/grading_result.py +37 -0
  34. marin_taskcompendium-0.2.160.dev38062449553/src/taskcompendium/harbor/__init__.py +4 -0
  35. marin_taskcompendium-0.2.160.dev38062449553/src/taskcompendium/harbor/compare.py +255 -0
  36. marin_taskcompendium-0.2.160.dev38062449553/src/taskcompendium/harbor/export.py +644 -0
  37. marin_taskcompendium-0.2.160.dev38062449553/src/taskcompendium/harbor/records.py +79 -0
  38. marin_taskcompendium-0.2.160.dev38062449553/src/taskcompendium/harbor/snapshots.py +122 -0
  39. marin_taskcompendium-0.2.160.dev38062449553/src/taskcompendium/harbor/tasktrove.py +72 -0
  40. marin_taskcompendium-0.2.160.dev38062449553/src/taskcompendium/importers/__init__.py +4 -0
  41. marin_taskcompendium-0.2.160.dev38062449553/src/taskcompendium/importers/nemo_predicted_action.py +200 -0
  42. marin_taskcompendium-0.2.160.dev38062449553/src/taskcompendium/importers/swe.py +78 -0
  43. marin_taskcompendium-0.2.160.dev38062449553/src/taskcompendium/importers/tasktrove/__init__.py +4 -0
  44. marin_taskcompendium-0.2.160.dev38062449553/src/taskcompendium/importers/tasktrove/convert.py +70 -0
  45. marin_taskcompendium-0.2.160.dev38062449553/src/taskcompendium/importers/tasktrove/mcqa.py +80 -0
  46. marin_taskcompendium-0.2.160.dev38062449553/src/taskcompendium/importers/tasktrove/models.py +30 -0
  47. marin_taskcompendium-0.2.160.dev38062449553/src/taskcompendium/models.py +891 -0
  48. marin_taskcompendium-0.2.160.dev38062449553/src/taskcompendium/pipeline/README.md +223 -0
  49. marin_taskcompendium-0.2.160.dev38062449553/src/taskcompendium/pipeline/__init__.py +4 -0
  50. marin_taskcompendium-0.2.160.dev38062449553/src/taskcompendium/pipeline/audit_schema.py +119 -0
  51. marin_taskcompendium-0.2.160.dev38062449553/src/taskcompendium/pipeline/chat_requests.py +100 -0
  52. marin_taskcompendium-0.2.160.dev38062449553/src/taskcompendium/pipeline/controls.py +270 -0
  53. marin_taskcompendium-0.2.160.dev38062449553/src/taskcompendium/pipeline/conversion.py +148 -0
  54. marin_taskcompendium-0.2.160.dev38062449553/src/taskcompendium/pipeline/environment_inventory.py +49 -0
  55. marin_taskcompendium-0.2.160.dev38062449553/src/taskcompendium/pipeline/execution_telemetry.py +141 -0
  56. marin_taskcompendium-0.2.160.dev38062449553/src/taskcompendium/pipeline/filtering.py +37 -0
  57. marin_taskcompendium-0.2.160.dev38062449553/src/taskcompendium/pipeline/fingerprints.py +110 -0
  58. marin_taskcompendium-0.2.160.dev38062449553/src/taskcompendium/pipeline/inputs.py +98 -0
  59. marin_taskcompendium-0.2.160.dev38062449553/src/taskcompendium/pipeline/models.py +325 -0
  60. marin_taskcompendium-0.2.160.dev38062449553/src/taskcompendium/pipeline/query_cache.py +259 -0
  61. marin_taskcompendium-0.2.160.dev38062449553/src/taskcompendium/pipeline/review.py +785 -0
  62. marin_taskcompendium-0.2.160.dev38062449553/src/taskcompendium/pipeline/review_requests.py +285 -0
  63. marin_taskcompendium-0.2.160.dev38062449553/src/taskcompendium/pipeline/sampling.py +44 -0
  64. marin_taskcompendium-0.2.160.dev38062449553/src/taskcompendium/pipeline/shard_outputs.py +58 -0
  65. marin_taskcompendium-0.2.160.dev38062449553/src/taskcompendium/pipeline/source_processing.py +1072 -0
  66. marin_taskcompendium-0.2.160.dev38062449553/src/taskcompendium/pipeline/source_quality.py +283 -0
  67. marin_taskcompendium-0.2.160.dev38062449553/src/taskcompendium/pipeline/source_verification.py +634 -0
  68. marin_taskcompendium-0.2.160.dev38062449553/src/taskcompendium/pipeline/sources.py +210 -0
  69. marin_taskcompendium-0.2.160.dev38062449553/src/taskcompendium/pipeline/stages.py +845 -0
  70. marin_taskcompendium-0.2.160.dev38062449553/src/taskcompendium/pipeline/transforms.py +221 -0
  71. marin_taskcompendium-0.2.160.dev38062449553/src/taskcompendium/pipeline/verification.py +192 -0
  72. marin_taskcompendium-0.2.160.dev38062449553/src/taskcompendium/runtime/__init__.py +7 -0
  73. marin_taskcompendium-0.2.160.dev38062449553/src/taskcompendium/runtime/environment.py +27 -0
  74. marin_taskcompendium-0.2.160.dev38062449553/src/taskcompendium/runtime/episode.py +98 -0
  75. marin_taskcompendium-0.2.160.dev38062449553/src/taskcompendium/runtime/grading.py +649 -0
  76. marin_taskcompendium-0.2.160.dev38062449553/src/taskcompendium/runtime/local.py +219 -0
  77. marin_taskcompendium-0.2.160.dev38062449553/src/taskcompendium/runtime/models.py +82 -0
  78. marin_taskcompendium-0.2.160.dev38062449553/src/taskcompendium/runtime/output_capture.py +113 -0
  79. marin_taskcompendium-0.2.160.dev38062449553/src/taskcompendium/runtime/resources.py +18 -0
  80. marin_taskcompendium-0.2.160.dev38062449553/src/taskcompendium/runtime/shell.py +286 -0
  81. marin_taskcompendium-0.2.160.dev38062449553/src/taskcompendium/runtime/task_grading.py +61 -0
  82. marin_taskcompendium-0.2.160.dev38062449553/src/taskcompendium/submission.py +188 -0
  83. marin_taskcompendium-0.2.160.dev38062449553/tests/conftest.py +13 -0
  84. marin_taskcompendium-0.2.160.dev38062449553/tests/fixtures/nemo/predicted-action.json +1 -0
  85. marin_taskcompendium-0.2.160.dev38062449553/tests/fixtures/nemo/predicted-action.provenance.json +11 -0
  86. marin_taskcompendium-0.2.160.dev38062449553/tests/fixtures/review/stack_pytest_long_evidence.json +12 -0
  87. marin_taskcompendium-0.2.160.dev38062449553/tests/fixtures/tasktrove/mcq-1961bdb52b5a.tar.gz +0 -0
  88. marin_taskcompendium-0.2.160.dev38062449553/tests/pipeline_stages.py +192 -0
  89. marin_taskcompendium-0.2.160.dev38062449553/tests/test_answer_formats.py +194 -0
  90. marin_taskcompendium-0.2.160.dev38062449553/tests/test_chat_requests.py +82 -0
  91. marin_taskcompendium-0.2.160.dev38062449553/tests/test_chat_review.py +199 -0
  92. marin_taskcompendium-0.2.160.dev38062449553/tests/test_controls.py +231 -0
  93. marin_taskcompendium-0.2.160.dev38062449553/tests/test_conversion.py +212 -0
  94. marin_taskcompendium-0.2.160.dev38062449553/tests/test_convert_answers.py +176 -0
  95. marin_taskcompendium-0.2.160.dev38062449553/tests/test_convert_executable.py +418 -0
  96. marin_taskcompendium-0.2.160.dev38062449553/tests/test_convert_json_schema.py +27 -0
  97. marin_taskcompendium-0.2.160.dev38062449553/tests/test_convert_nemotron_ultra.py +114 -0
  98. marin_taskcompendium-0.2.160.dev38062449553/tests/test_convert_preference.py +84 -0
  99. marin_taskcompendium-0.2.160.dev38062449553/tests/test_convert_script_grader.py +70 -0
  100. marin_taskcompendium-0.2.160.dev38062449553/tests/test_convert_tasktrove.py +89 -0
  101. marin_taskcompendium-0.2.160.dev38062449553/tests/test_environment.py +23 -0
  102. marin_taskcompendium-0.2.160.dev38062449553/tests/test_execution_telemetry.py +99 -0
  103. marin_taskcompendium-0.2.160.dev38062449553/tests/test_grader_package.py +206 -0
  104. marin_taskcompendium-0.2.160.dev38062449553/tests/test_harbor_compare.py +238 -0
  105. marin_taskcompendium-0.2.160.dev38062449553/tests/test_harbor_export.py +65 -0
  106. marin_taskcompendium-0.2.160.dev38062449553/tests/test_local_runtime.py +27 -0
  107. marin_taskcompendium-0.2.160.dev38062449553/tests/test_multiple_choice_verifier.py +47 -0
  108. marin_taskcompendium-0.2.160.dev38062449553/tests/test_nemo_predicted_action.py +319 -0
  109. marin_taskcompendium-0.2.160.dev38062449553/tests/test_output_capture.py +263 -0
  110. marin_taskcompendium-0.2.160.dev38062449553/tests/test_pipeline.py +1319 -0
  111. marin_taskcompendium-0.2.160.dev38062449553/tests/test_review_evidence.py +147 -0
  112. marin_taskcompendium-0.2.160.dev38062449553/tests/test_runtime.py +428 -0
  113. marin_taskcompendium-0.2.160.dev38062449553/tests/test_schema.py +318 -0
  114. marin_taskcompendium-0.2.160.dev38062449553/tests/test_script_grader.py +551 -0
  115. marin_taskcompendium-0.2.160.dev38062449553/tests/test_source_processing.py +897 -0
  116. marin_taskcompendium-0.2.160.dev38062449553/tests/test_source_quality.py +395 -0
  117. marin_taskcompendium-0.2.160.dev38062449553/tests/test_source_verification.py +560 -0
  118. marin_taskcompendium-0.2.160.dev38062449553/tests/test_submission_grading.py +191 -0
  119. marin_taskcompendium-0.2.160.dev38062449553/tests/test_tasktrove_answers.py +173 -0
@@ -0,0 +1,253 @@
1
+ # redundant, but Ray looks for this otherwise.
2
+ .git
3
+
4
+ logs/
5
+
6
+ # CPU profiles
7
+ prof/
8
+
9
+ # Downloaded build tools (zig, etc.)
10
+ .tools/
11
+
12
+ tests/snapshots/outputs
13
+ tests/snapshots/diffs
14
+
15
+ # don't log data/MD outputs to git
16
+ data/*
17
+ output/*
18
+ outputs/*
19
+
20
+ # Snapshot diffs and outputs
21
+ tests/snapshots/*/outputs/*
22
+ tests/snapshots/*/diffs/*
23
+
24
+ # This is mainly for Ray and using submodule
25
+ */**/.git
26
+
27
+ ### Python template
28
+ # Byte-compiled / optimized / DLL files
29
+ __pycache__/
30
+ *.py[cod]
31
+ *$py.class
32
+
33
+ # C extensions
34
+ *.so
35
+
36
+ # pypa/gh-action-pypi-publish caches its Docker action manifest here.
37
+ .github/.tmp/
38
+
39
+ # Distribution / packaging
40
+ .Python
41
+ build/
42
+ develop-eggs/
43
+ dist/
44
+ downloads/
45
+ eggs/
46
+ .eggs/
47
+ lib64/
48
+ parts/
49
+ sdist/
50
+ local_store/
51
+ wheels/
52
+ share/python-wheels/
53
+ *.egg-info/
54
+ .installed.cfg
55
+ *.egg
56
+ MANIFEST
57
+
58
+ # PyInstaller
59
+ # Usually these files are written by a python script from a template
60
+ # before PyInstaller builds the exe, so as to inject date/other infos into it.
61
+ *.manifest
62
+ *.spec
63
+
64
+ # Installer logs
65
+ pip-log.txt
66
+ pip-delete-this-directory.txt
67
+
68
+ # Unit test / coverage reports
69
+ htmlcov/
70
+ .tox/
71
+ .nox/
72
+ .coverage
73
+ .coverage.*
74
+ .cache
75
+ nosetests.xml
76
+ coverage.xml
77
+ *.cover
78
+ *.py,cover
79
+ .hypothesis/
80
+ .pytest_cache/
81
+ cover/
82
+
83
+ # Translations
84
+ *.mo
85
+ *.pot
86
+
87
+ # Django stuff:
88
+ *.log
89
+ local_settings.py
90
+ db.sqlite3
91
+ db.sqlite3-journal
92
+
93
+ # Flask stuff:
94
+ instance/
95
+ .webassets-cache
96
+
97
+ # Scrapy stuff:
98
+ .scrapy
99
+
100
+ # Sphinx documentation
101
+ docs/_build/
102
+
103
+ # PyBuilder
104
+ .pybuilder/
105
+ target/
106
+
107
+ # Jupyter Notebook
108
+ .ipynb_checkpoints
109
+
110
+ # IPython
111
+ profile_default/
112
+ ipython_config.py
113
+
114
+ # pyenv
115
+ # For a library or package, you might want to ignore these files since the code is
116
+ # intended to run in multiple environments; otherwise, check them in:
117
+ # .python-version
118
+
119
+ # pipenv
120
+ # According to pypa/pipenv#598, it is recommended to include Pipfile.lock in version control.
121
+ # However, in case of collaboration, if having platform-specific dependencies or dependencies
122
+ # having no cross-platform support, pipenv may install dependencies that don't work, or not
123
+ # install all needed dependencies.
124
+ #Pipfile.lock
125
+
126
+ # poetry
127
+ # Similar to Pipfile.lock, it is generally recommended to include poetry.lock in version control.
128
+ # This is especially recommended for binary packages to ensure reproducibility, and is more
129
+ # commonly ignored for libraries.
130
+ # https://python-poetry.org/docs/basic-usage/#commit-your-poetrylock-file-to-version-control
131
+ #poetry.lock
132
+
133
+ # pdm
134
+ # Similar to Pipfile.lock, it is generally recommended to include pdm.lock in version control.
135
+ #pdm.lock
136
+ # pdm stores project-wide configurations in .pdm.toml, but it is recommended to not include it
137
+ # in version control.
138
+ # https://pdm.fming.dev/#use-with-ide
139
+ .pdm.toml
140
+
141
+ # PEP 582; used by e.g. github.com/David-OConnor/pyflow and github.com/pdm-project/pdm
142
+ __pypackages__/
143
+
144
+ # Celery stuff
145
+ celerybeat-schedule
146
+ celerybeat.pid
147
+
148
+ # SageMath parsed files
149
+ *.sage.py
150
+
151
+ # Environments
152
+ .env
153
+ .venv
154
+ env/
155
+ venv/
156
+ ENV/
157
+ env.bak/
158
+ venv.bak/
159
+
160
+ # Spyder project settings
161
+ .spyderproject
162
+ .spyproject
163
+
164
+ # Rope project settings
165
+ .ropeproject
166
+
167
+ # mkdocs documentation
168
+ /site
169
+
170
+ # mypy
171
+ .mypy_cache/
172
+ .dmypy.json
173
+ dmypy.json
174
+
175
+ # Pyre type checker
176
+ .pyre/
177
+
178
+ # Ruff
179
+ .ruff_cache/
180
+
181
+ # pytype static type analyzer
182
+ .pytype/
183
+
184
+ # Cython debug symbols
185
+ cython_debug/
186
+
187
+ # PyCharm
188
+ # JetBrains specific template is maintained in a separate JetBrains.gitignore that can
189
+ # be found at https://github.com/github/gitignore/blob/main/Global/JetBrains.gitignore
190
+ # and can be added to the global gitignore or merged into this file. For a more nuclear
191
+ # option (not recommended) you can uncomment the following to ignore the entire idea folder.
192
+ .idea/
193
+ *.iml
194
+
195
+ # IDE Config
196
+ .vscode/
197
+
198
+ # Mac OS
199
+ .DS_Store
200
+
201
+ # Secrets
202
+ credentials.json
203
+ marin/crawl/bigquery-gcs-key.json
204
+
205
+ # Archive
206
+ archive/
207
+
208
+ # Caches and Outputs
209
+ !/scripts/web/output/
210
+ !/output/
211
+
212
+ # csv
213
+ *.csv
214
+
215
+ # wandb logs
216
+ wandb
217
+ artifacts
218
+
219
+ # Ignore generated credentials from google-github-actions/auth
220
+ gha-creds-*.json
221
+
222
+ .aider*
223
+ .git/*
224
+
225
+ *.jsonl
226
+ **/*.jsonl
227
+ # Echo's judged search corpus is source, not a run artifact.
228
+ !infra/marina/apps/echo/benchmarks/*.jsonl
229
+ scr/*
230
+ .weaver/
231
+
232
+ # Local host Marin config
233
+ .marin.yaml
234
+
235
+ /scratch
236
+
237
+ .forge
238
+ .claude/*
239
+ !.claude/skills
240
+ !.claude/agents
241
+ !.claude/settings.json
242
+ .agents/tmp/
243
+ .codex
244
+ .entire
245
+ .beads
246
+
247
+ .worktrees
248
+ .obsidian
249
+ .cw_env
250
+ .localbin/
251
+
252
+ # Stamped into built artifacts by lib/iris/hatch_build.py; never tracked.
253
+ lib/iris/src/iris/_build_info.py
@@ -0,0 +1,16 @@
1
+ Metadata-Version: 2.5
2
+ Name: marin-taskcompendium
3
+ Version: 0.2.160.dev38062449553
4
+ Summary: Semantic task contracts, submission extraction, and shared grading
5
+ Requires-Python: <3.14,>=3.12
6
+ Requires-Dist: marin-rigging==0.2.160.dev38062449553
7
+ Requires-Dist: marin-shellbox==0.2.160.dev38062449553
8
+ Requires-Dist: marin-verifyit[judge,math,reasoning-gym,schema]==0.2.160.dev38062449553
9
+ Requires-Dist: pydantic>=2.0
10
+ Requires-Dist: tomlkit>=0.13
11
+ Provides-Extra: pipeline
12
+ Requires-Dist: datasets<5.0.0,>=3.1.0; extra == 'pipeline'
13
+ Requires-Dist: marin-finestore==0.2.160.dev38062449553; extra == 'pipeline'
14
+ Requires-Dist: marin-zephyr==0.2.160.dev38062449553; extra == 'pipeline'
15
+ Requires-Dist: numpy>=1.26; extra == 'pipeline'
16
+ Requires-Dist: pyarrow>=15.0; extra == 'pipeline'
@@ -0,0 +1,356 @@
1
+ # TaskCompendium
2
+
3
+ The [task curation pipeline](../../docs/references/task-curation.md) downloads pinned sources, normalizes tasks, runs grading checks and GLM review, then writes final filtering decisions to sharded Parquet. Its audit retains every selected input, source locator, edit and rejection reason. Dataset declarations in the experiment name the pinned source, converter, rubric and grader controls; `taskcompendium.convert` holds the conversion techniques they share.
4
+
5
+ For ingestion work, start with the [pipeline overview](src/taskcompendium/pipeline/README.md)
6
+ and the [experiment flow](../../experiments/post_training/task_curation/README.md).
7
+ The sections below describe the task model and its presentation and grading contracts.
8
+
9
+ ## What problem does it solve?
10
+
11
+ TaskCompendium stores a task's semantic contract, extracts one final submission with the task's answer format, and grades it with the task's grader. Expected answers, grader configuration, and verifier resources stay out of the model request.
12
+
13
+ Actor execution, model requests, tool dispatch, and actor environment lifecycle belong to the caller’s runtime. `taskcompendium.runtime` grades in a fresh Shellbox machine when the grader needs one. RolloutEngine provides the actor runtime; see [task rollouts](../../docs/references/task-rollouts.md). TaskCompendium does not depend on RolloutEngine.
14
+
15
+ ## What does it contain?
16
+
17
+ - **Task specs** describe source problems, required capabilities, final results, answer formats, and graders.
18
+ - **Answer formats** describe how to request and extract a result, such as plain text, a JSON answer, or a final function call.
19
+ - **Typed evidence** records the conversation and an acquired text, JSON, action, or environment-state submission.
20
+ - **Grading** scores the extracted evidence in process or in a fresh grading machine, or reports that the task cannot be graded here.
21
+
22
+ ```mermaid
23
+ flowchart LR
24
+ S[TaskSpec with answer format and grader] --> G[Extract once and grade]
25
+ A[Attempt evidence from runtime] --> G
26
+ G --> R[Typed grade result]
27
+ ```
28
+
29
+ ## What is a task spec?
30
+
31
+ `TaskSpec` defines one source task, including its grader and expected values. An importer or author creates it before choosing an execution runtime.
32
+
33
+ | Field | Meaning |
34
+ | --- | --- |
35
+ | `id` | Stable identity for this task. |
36
+ | `context` | The ordered model-visible conversation: text messages, historical assistant function calls, and tool results. |
37
+ | `environment_requirements` | Required capabilities, pinned initial workspace, and named tool-provider contracts. |
38
+ | `final_tools` | An ordered list of functions that terminate a chat. They are not backed by a tool provider. |
39
+ | `interaction_tools` | Executable function declarations used by the optional episode runtime. |
40
+ | `output_paths` | Absolute output paths captured by the optional episode runtime, outside `/tests` and `/logs/verifier`. |
41
+ | `output_directories` | Workspace roots, relative fnmatch patterns and explicit file-count/aggregate-byte capture budgets. |
42
+ | `answer_type` | The semantic result: `text`, `number`, `json`, `file`, `state`, `workspace_state`, or `native_action`. |
43
+ | `answer_format` | How the final answer is requested from the model and extracted. See [Answer formats](#answer-formats). |
44
+ | `grader` | How an attempt is graded. See [What is a grader?](#what-is-a-grader) |
45
+ | `source` | Upstream dataset, revision, row, and importer revision retained as audit provenance. |
46
+ | `schema_version` | Version of the serialized spec: `0.25`. Readers reject other versions. |
47
+ | `resources` | Inline files grouped under `all`, `worker`, `oracle`, and `verifier` visibility. |
48
+ | `tags` | Arbitrary descriptive strings, retained in order, including duplicates and empty strings. |
49
+
50
+ A task has one final result. TaskSpec does not define ordered task stages or stage-reward aggregation.
51
+
52
+ Directory capture requires the actor's `python3` capability and a real POSIX
53
+ Python interpreter; ShellSim does not support it. Selection patterns use
54
+ case-sensitive fnmatch semantics, where `*` includes `/` and hidden paths.
55
+ Capture preserves unsorted depth-first filesystem order, skips symlinks in
56
+ directory selections, and fails the whole capture on file or byte overflow.
57
+ Existing named `output_paths` retain their file/symlink behavior. Roots must stay
58
+ inside the grader's workspace and outside `/tests`, `/logs/verifier` and
59
+ `/solution`. Membership and budgets are checked again before files are staged for grading.
60
+
61
+ `context.events` is the model-visible conversation prefix. A text event retains its role and content. Historical assistant calls and tool results retain their call IDs and order; a runtime preserves this history when presenting the task to the model. `answer_type` does not prescribe a wrapper such as JSON; `answer_format` does.
62
+
63
+ `ConversationInput` is not a lossless Responses API transcript. It excludes provider reasoning items. The NeMo importer omits an unencrypted historical reasoning summary only when a visible assistant message or function call follows it. It rejects encrypted reasoning and reasoning left at the decision point. Exact provider continuation from reasoning state is outside this contract.
64
+
65
+ An **answer** is the task’s semantic result, identified by `answer_type`. A **submission** is the typed value the task's answer format extracts from a completed attempt for grading. Different answer formats can yield the same submission and carry the same answer.
66
+
67
+ For example, a task asking “What is 7 + 5?” can have `answer_type=number` and a `numeric` grader that expects `12`. The `PlainText`, `JsonAnswer`, and `AnswerCall` formats carry that answer as `12`, `{"answer":"12"}`, and a final `submit_answer({"answer":"12"})` call, respectively. Each extracts the string `12` for the same grader, which accepts both `12` and `12.0`. A task asking for a function call has `answer_type=native_action`; its context retains the conversation and its `final_tools` field describes the source functions. The grader and expected answer are never added to the model-visible input. Importers must make source output instructions agree with the task's answer format, or reject rows they cannot safely rewrite. A raw-output requirement in a source message would conflict with `JsonAnswer` or `AnswerCall`; `answer_type=text` alone cannot detect that conflict in prose.
68
+
69
+ ## What can a task represent?
70
+
71
+ ### Text answers
72
+
73
+ A text task uses `answer_type=text`. `PlainText`, `Boxed`, `JsonAnswer`, and `AnswerCall` can carry its answer. The `exact` mode compares normalized text; `mcq` grades a single option letter. Invalid MCQ option text scores zero through verifyit. Both modes grade the answer extracted by any of these formats.
74
+
75
+ ### Numeric answers
76
+
77
+ A numeric task uses `answer_type=number` and can use the same answer formats as text. Expected values are required numeric literal strings, preserving integers, decimals and fractions exactly. Absolute and relative tolerances are required finite nonnegative floats. The `numeric` mode reads the last boxed answer, or exactly one numeric literal from the last nonempty line when no box is present. Surrounding prose, including negation, is ignored. Missing, malformed, or ambiguous numeric output is `submission_failure`; a valid wrong number is `graded` with reward `0.0`.
78
+
79
+ `taskcompendium.grading.grade_result` maps verifyit rewards to `GradeResult` the same way for in-process grading and for the verdict the verifyit command writes in a grading machine. Malformed numeric text is `submission_failure` with reward `0.0` in both.
80
+
81
+ ### Final function calls
82
+
83
+ A task whose result is a function call uses `answer_type=native_action` and the `FinalAction` answer format. Its `final_tools` field declares the available functions. `FinalAction` defines whether a call is required and the maximum call count. It extracts the assistant's final message, and the `predicted_action` mode compares the submitted function names and decoded argument objects with the expected calls. Grading compares the submitted calls without dispatching them.
84
+
85
+ For text and numeric tasks, the `AnswerCall` format adds `submit_answer(answer: string)` alongside the task’s final tools. It extracts the call's `answer` argument and passes it to the task's grader. A non-call response or a call to another function receives `submission_failure` with reward `0.0`.
86
+
87
+ ### Files and state
88
+
89
+ `answer_type=file` names a file result. `answer_type=workspace_state` names the final filesystem workspace. `answer_type=state` names arbitrary resulting environment state, including provider state outside a filesystem. Acquiring these results requires an actor runtime. These tasks still declare an `answer_format`, but grading does not read it. `file` and `workspace_state` answers require a grader with an environment. In-process grading uses supplied evidence; the optional TaskCompendium episode runtime captures declared output files. The `structured_exact` mode compares JSON values and ordered arrays. Numbers compare by value by default (`16` equals `16.0`); booleans remain distinct. Set `numeric_types="strict"` to require exact numeric scalar types.
90
+
91
+ `answer_type=json` is a JSON answer from the model, independent of environment state. `JsonValueAnswer` parses the complete final chat text into a `JsonSubmission` for `structured_exact`. `StateSubmission` is evidence acquired from an environment by its runtime. `structured_exact` accepts either. The `JsonAnswer` format instead unwraps an `answer` string for text or numeric tasks. Both formats reject duplicate keys at every nesting level and nonfinite numbers. Tool-call arguments are decoded before entering the conversation evidence. Runtimes own provider-response decoding and may report malformed tool-call arguments as a structural submission failure with reward `0.0`. Both historical and final calls contain typed argument objects.
92
+
93
+ Public expectations belong in `context`: for example, the columns a CSV must contain or the behavior a repaired project must provide. The grader checks those expectations. The answer format chooses how the result is delivered and extracted. `answer_type` identifies the answer’s semantic kind.
94
+
95
+ ## Environment requirements
96
+
97
+ `environment_requirements` declares the initial state and operations needed to solve a task.
98
+
99
+ `compatible_backends` lists the Shellbox backends the source author permits for
100
+ this environment, such as `shellsim` or `gvisor`. A grader environment has its own
101
+ list. Empty means no Shellbox backend is declared; it is not a wildcard.
102
+ TaskCompendium's shell runtime and `grade_task` check the selected factory against
103
+ the appropriate list before creating a machine. A required Docker image excludes ShellSim. See the
104
+ [source-author rubric](src/taskcompendium/pipeline/README.md) for compatibility
105
+ criteria and the distinction between declarations and sampled runtime evidence.
106
+
107
+ | Field | Meaning |
108
+ | --- | --- |
109
+ | `capabilities` | Unique operation names, such as `shell`, `network`, `filesystem`, `process`, or `browser`. Names are open so future capabilities can be represented. |
110
+ | `compatible_backends` | Unique Shellbox backend names permitted by the source author; empty declares none. |
111
+ | `docker_image` | Optional immutable image reference, such as `registry/project@sha256:<64 lowercase hex digits>`. Tags alone are rejected. |
112
+ | `docker_build` | Optional `DockerBuildContext(files=...)` containing inline regular files relative to the build-context root, including `Dockerfile`. The recipe is unresolved data. |
113
+ | `working_directory` | Optional normalized absolute POSIX path for the main workspace. Omission declares no required working directory. |
114
+ | `setup_commands` | Ordered commands required to establish the initial workspace. |
115
+ | `environment_variables` | String values required in the worker or grading machine environment. |
116
+ | `tool_providers` | Mapping from a task-local provider instance name to a required action interface and initial state. |
117
+ | `packages_lock` | Storage URL of a uv-compiled, hash-pinned requirements lock. A `local` environment requires it: the host builds the lock into a Python environment for the commands it runs. Other environments must omit it. |
118
+
119
+ Each `ProviderRequirement` contains `action_interface`, a versioned contract name such as `workplace:v1`, and required `initial_state`, a JSON value such as a string, null, or an object. Two named instances can require the same interface with different initial states. No digest is required. The selected runtime owns provider implementation, transport, state initialization, reset, and tool execution. `final_tools` contains only ordered function definitions advertised at the final decision point; it supplies no implementation.
120
+
121
+ The task's environment and worker file mounts describe worker initial state. A grader that runs in its own machine declares that machine's image or build recipe, capabilities, setup commands, and environment variables in `grader.environment`. A grader environment requires a `docker_image`, `docker_build`, or local `packages_lock`, and cannot declare tool providers. Runtimes keep those requirements and verifier resources separate from the worker environment.
122
+
123
+ `docker_image`, `docker_build`, and `packages_lock` are mutually exclusive.
124
+ Build files preserve bytes, modes and timestamps; paths cannot collide. Resource
125
+ budgets count actor and grader contexts separately. Execution requires the caller
126
+ to build each context and replace it with a digest-pinned image and supported
127
+ backends. TaskCompendium supplies no build resolver; runtimes and SAMPLE/FULL
128
+ controls reject unresolved contexts. Local and ShellSim backends cannot declare them.
129
+
130
+ The [Harbor exporter](src/taskcompendium/harbor/export.py) emits build inputs;
131
+ see the [campaign quickstart](../../experiments/post_training/task_curation/README.md#harbor-compatibility-view)
132
+ for execution limits. The Harbor-to-TaskSpec importer requires a prebuilt image.
133
+
134
+ ## Resource mounts
135
+
136
+ `resources` is a `ResourceGroups` object. Each group contains an ordered list of `TaskResource` mounts.
137
+
138
+ | Group | Visibility |
139
+ | --- | --- |
140
+ | `all` | Shared inputs visible to the worker, oracle, and grader. |
141
+ | `worker` | Model-visible inputs. |
142
+ | `oracle` | Reference material for oracle controls, hidden from the model. |
143
+ | `verifier` | Grader inputs, hidden from the model. |
144
+
145
+ Gold answers and hidden tests belong in `oracle` or `verifier`. An `all` resource is model-visible. A role receives `all` followed by its own mounts. Destinations must be distinct in that combined sequence and have no file/directory ancestor collisions. Names retain POSIX case distinctions. Separate role-specific groups can reuse a relative path without sharing their content.
146
+
147
+ Oracle resources are reserved for trusted reference-solution generation. Verifier resources are used when evaluating a candidate result. These groups declare access; they do not require an oracle or evaluator process to run.
148
+
149
+ The TaskCompendium and RolloutEngine runtimes install shared, worker, and oracle resources at `/<path>`: a resource at `app/project/input.txt` appears at `/app/project/input.txt`. Worker mounts are established before setup commands run. A grading machine receives verifier resources at `/tests/<path>`; in-process modes read them by that relative path.
150
+
151
+ Each `TaskResource` contains one inline file, with these fields:
152
+
153
+ | Field | Meaning |
154
+ | --- | --- |
155
+ | `path` | Normalized relative destination: `/<path>` for shared, worker, and oracle resources, and `/tests/<path>` for verifier resources. |
156
+ | `source` | An `InlineFile` containing the exact file bytes, encoded as canonical base64. |
157
+ | `mode` | Optional Unix permission mode as a three- or four-digit octal string, such as `0644` or `0755`. An explicit mode applies to the mounted file. Files default to `0644` when the mode is omitted. |
158
+ | `mtime_ns` | Optional integer Unix modification timestamp in nanoseconds for the mounted file. Omission leaves the timestamp unspecified. |
159
+
160
+ `InlineFile` has `kind="inline_file"` and `content_base64`. UTF-8 text uses the same byte representation as binary files. Resource contents are stored in the task spec; no dataset root or process working directory is used to locate them. Shared external files are deferred until a `TaskSet` contract defines their location and loading.
161
+
162
+ An archive can be included as ordinary file bytes, but resource mounting does not extract it. Paths may contain directories to locate the file within the workspace. Directory resources and recursive copies are unsupported. Schema decoding performs no filesystem inspection or I/O.
163
+
164
+ Grouped inline resources can contain:
165
+
166
+ ```json
167
+ {
168
+ "resources": {
169
+ "all": [
170
+ {
171
+ "path": "README.txt",
172
+ "source": {"kind": "inline_file", "content_base64": "VXNlIHRoZSBzdXBwbGllZCBwcm9qZWN0Lg=="},
173
+ "mode": "0444",
174
+ "mtime_ns": 1725555600000000000
175
+ }
176
+ ],
177
+ "worker": [
178
+ {"path": "project/input.txt", "source": {"kind": "inline_file", "content_base64": "cHVibGljIGlucHV0"}}
179
+ ],
180
+ "oracle": [
181
+ {"path": "answer.txt", "source": {"kind": "inline_file", "content_base64": "cHJpdmF0ZSByZWZlcmVuY2U="}}
182
+ ],
183
+ "verifier": [
184
+ {"path": "checks/grade.py", "source": {"kind": "inline_file", "content_base64": "cHJpdmF0ZSBjaGVja3M="}}
185
+ ]
186
+ }
187
+ }
188
+ ```
189
+
190
+ File materializers must reject unsafe destinations and collisions, and enforce byte limits. The schema and in-process grading do not materialize resources. Execution runtimes supply mounts and role isolation; `grade_in_sandbox` stages grading inputs into a fresh machine.
191
+
192
+ ## What can we import?
193
+
194
+ ### TaskTrove MCQA
195
+
196
+ The TaskTrove MCQA importer reads archives from a cleaned release. See the [published TaskTrove Clean dataset](https://huggingface.co/datasets/open-athena/task-trove). Its caller passes the archive bytes, upstream subset, archive path, and release provenance to `read_archive`. The reader checks the subset and path against the archive manifest; the release URI and revision are caller-supplied provenance. The importer checks the source answer-line template before replacing it with a one-letter instruction. Its text answer uses the `PlainText` format. The `mcq` grader stores the expected letter and option count and grades in process with the `verifyit` MCQ scorer. This importer supports only MCQ mode. Executable TaskTrove modes need verifier resources and a grading machine.
197
+
198
+ ### NeMo predicted function calls
199
+
200
+ `taskcompendium.importers.nemo_predicted_action.import_row` accepts a NeMo predicted-function-call row and a caller-pinned digest of that row. `canonical_sha256(row)` hashes its UTF-8 JSON with sorted keys and compact separators; record the digest with the source revision before importing. The importer returns a TaskSpec with `answer_type=native_action` and a `FinalAction` answer format. The context carries the source conversation; `final_tools` carries advertised functions. `FinalAction.require_call` is set when the source requires a tool call, and `FinalAction.max_calls=1` when the source disables parallel calls. The expected function calls stay in the `predicted_action` grader. There is one stored conversation, with no second flattened prompt to keep in sync.
201
+
202
+ The runtime presents the source turns and function definitions and retains the final response as typed evidence. The grader compares submitted function names and JSON arguments. The importer rejects rows whose expected action is an assistant text message because the source comparator gives any message full credit; it also rejects request settings it cannot carry. The pinned fixture records the NeMo Gym repository revision and blob SHA in `tests/fixtures/nemo/predicted-action.provenance.json`. Numeric tolerance is used only when explicitly set in the grader's parameters.
203
+
204
+ ## What is a grader?
205
+
206
+ `TaskSpec.grader` declares how an attempt is graded. Its `kind` field selects one of four classes in `taskcompendium.models`:
207
+
208
+ | `kind` | Class | Grading |
209
+ | --- | --- | --- |
210
+ | `verifyit` | `VerifyitGrader` | A stock verifyit `mode` and its `parameters`: the keys of verifyit's `verifier.toml` table other than `mode`. |
211
+ | `script` | `ScriptGrader` | A command run after the episode in a fresh machine built from a pinned image. |
212
+ | `session` | `SessionGrader` | A registered custom RolloutEngine `TaskSession` grades the task. |
213
+ | `none` | `NoGrader` | The task cannot be graded here. `reason` says why; `contract` optionally keeps the source's grading terms. |
214
+
215
+ A `VerifyitGrader` without an `environment` grades the extracted answer in process. Its mode must be one of `verifyit.candidate.IN_PROCESS_MODES`: `exact`, `numeric`, `mcq`, `math`, `ifeval`, `json-schema`, `xml-elements`, `csv-columns`, `structured_exact`, or `predicted_action`. A `VerifyitGrader` with an `environment` runs the verifyit command in a fresh machine built from that environment, with verifier resources under `/tests` and the verifyit specification at `/tests/verifier.toml`. Modes that execute code or call a model, such as `pytest`, `stdio`, and `judge`, grade only there. Construction validates `parameters` with verifyit.
216
+
217
+ A `ScriptGrader` declares:
218
+
219
+ | Field | Meaning |
220
+ | --- | --- |
221
+ | `argv`, `cwd`, `env` | The grader command, its working directory (default `/app`), and added environment variables. |
222
+ | `environment` | The grading machine's requirements. |
223
+ | `collect` | Commands run as root on the agent's machine before grading. |
224
+ | `artifacts` | Files and directories copied from the agent's machine into the grading machine, with exclusions and a missing-file policy. |
225
+ | `answer_path` | Where the extracted answer is written (default `/app/answer.txt`): text as text, JSON as JSON, and a final action as the final assistant message's JSON. |
226
+ | `conversation_path` | Where the conversation is written as OpenAI-style chat messages in JSON (default `/tests/conversation.json`). |
227
+ | `reward` | `StdoutReward` (default), `ExitCodeReward`, or `FileReward`. |
228
+ | `timeout` | Deadline in seconds for each grading command (default 600). |
229
+
230
+ `StdoutReward` requires exit code 0 and one finite number on standard output; anything else is `infra_error`. `ExitCodeReward` scores 1.0 for exit code 0 and 0.0 otherwise. `FileReward.files` lists candidate reward files in priority order; the first that exists supplies the score. Each `RewardFile` holds a number, or a JSON object whose `key` (default `reward`) holds the score and whose optional `detail` object becomes the grade detail. `FileReward.pass_above` sets `GradeResult.passed`; `ExitCodeReward` sets it from the exit code. A missing, empty, or invalid reward file is `infra_error`. A timed-out command is `infra_error` for every reward source.
231
+
232
+ TaskSpec validation ties the grader to the answer:
233
+
234
+ - An in-process `VerifyitGrader` cannot grade `file` or `workspace_state` answers.
235
+ - A `VerifyitGrader` with an environment cannot grade `native_action` answers. Its workspace modes (`stdio`, `pytest`, `junit`, and `gotest`) cannot grade `text`, `number`, or `json` answers.
236
+ - `ScriptGrader.answer_path` must be `None` for `file`, `state`, and `workspace_state` answers. `conversation_path` must not replace a verifier resource.
237
+ - Answer paths, output paths, and output directories must lie outside `/tests` and `/logs/verifier`.
238
+
239
+ A `SessionGrader` task is graded by the registered custom `TaskSession` that runs it; TaskCompendium cannot grade it. A `NoGrader` task grades as `unavailable`, with no reward and its `reason` as the error. Converters use `NoGrader` when the source evaluator needs a runtime this repository cannot provide; its `contract` records the evaluator, source revision, grading data and runtime requirements. `taskcompendium.grader.grader_config(task)` returns a copy of a `NoGrader`'s contract, or the `config.json` verifier resource of another grader.
240
+
241
+ `taskcompendium.grader.GraderPackage` pairs a grader with its verifier resources, whose paths are relative to `/tests`. `verifyit_package(spec, resources=(), environment=None)` builds a `VerifyitGrader` from a verifyit `Spec`. Build other packages directly with `GraderPackage(grader, resources)`.
242
+
243
+ ```python
244
+ from taskcompendium.grader import verifyit_package
245
+ from verifyit.spec import ExactSpec, FunctionCall, McqSpec, PredictedActionSpec, StructuredExactSpec
246
+
247
+ text_grader = verifyit_package(ExactSpec(expected=("expected text",))).grader
248
+ mcq_grader = verifyit_package(McqSpec(expected="C", options=4)).grader
249
+ json_grader = verifyit_package(StructuredExactSpec(expected={"value": 16})).grader
250
+ action_grader = verifyit_package(
251
+ PredictedActionSpec(expected_calls=(FunctionCall(name="lookup", arguments={"city": "Paris"}),))
252
+ ).grader
253
+ ```
254
+
255
+ A grader scores one acquired answer. Comparative scoring across several attempts, cohort membership, and grading phase belong to the trainer. Standard grading modes and in-process scoring live in `verifyit`; TaskCompendium has no grader registry or separate standard verifier schema.
256
+
257
+ ## Answer formats
258
+
259
+ `TaskSpec.answer_format` says how the final answer is requested from the model and extracted from `conversation.events[-1]`, the final assistant message or call batch. Earlier events are validated history; historical calls must have their results before the final submission.
260
+
261
+ | Format | `kind` | Extracts | Answer types |
262
+ | --- | --- | --- | --- |
263
+ | `PlainText` | `plain_text` | The whole final message | `text`, `number` |
264
+ | `Boxed` | `boxed` | The content of the last `\boxed{...}`, else the whole final message | `text`, `number` |
265
+ | `JsonAnswer` | `json_answer` | The nonempty string `answer` field of a JSON object | `text`, `number` |
266
+ | `JsonValueAnswer` | `json_value` | The whole final message as one JSON value | `json` |
267
+ | `AnswerCall` | `answer_call` | The `answer` argument of one `submit_answer` call | `text`, `number` |
268
+ | `FinalAction` | `final_action` | The final message: calls to `final_tools`, or text | `native_action` |
269
+
270
+ A `text`, `number`, `json`, or `native_action` task must use a format that carries its answer type. `submission_compatibility(task)` also checks that the grader's verifyit mode accepts the submission the format extracts (`TextSubmission`, `JsonSubmission`, or `ActionSubmission`), that a `FinalAction` task has final tools, and that no final tool is named `submit_answer` under `AnswerCall`.
271
+
272
+ Answer formats preserve the task's advertised functions. `AnswerCall` adds `submit_answer` with an answer string. `FinalAction(require_call=True)` requires a call, and `FinalAction(max_calls=1)` limits the submission to one call. Missing required calls and excessive call counts are `submission_failure` with reward `0.0`.
273
+
274
+ `taskcompendium.submission` builds model requests from the task alone. `submission_instruction(answer_format)` returns the format's final-answer instruction, and `answer_call_tool()` returns the `submit_answer` tool definition. `chat_request(task)` and `render_instruction(task)` combine them with the public conversation and final tools. These helpers perform no execution.
275
+
276
+ ## Grading
277
+
278
+ Create a task:
279
+
280
+ ```python
281
+ from taskcompendium.grader import verifyit_package
282
+ from taskcompendium.models import (
283
+ AnswerType,
284
+ ConversationInput,
285
+ EnvironmentRequirements,
286
+ PlainText,
287
+ Source,
288
+ TaskSpec,
289
+ TextMessage,
290
+ )
291
+ from verifyit.spec import NumericSpec
292
+
293
+ spec = TaskSpec(
294
+ id="arithmetic-7-plus-5",
295
+ context=ConversationInput(events=(TextMessage(role="user", content="What is 7 + 5?"),)),
296
+ environment_requirements=EnvironmentRequirements(),
297
+ answer_type=AnswerType.NUMBER,
298
+ answer_format=PlainText(),
299
+ grader=verifyit_package(NumericSpec(expected="12", tolerance_abs=0.0, tolerance_rel=0.0)).grader,
300
+ source=Source(dataset="hand-authored", revision="2026-09-16", row="arithmetic-7-plus-5", importer_revision="1"),
301
+ )
302
+ ```
303
+
304
+ After execution, the caller supplies the complete conversation to `GradingAttempt`:
305
+
306
+ ```python
307
+ from taskcompendium.grading import grade_answer
308
+ from taskcompendium.models import ConversationTrace, GradingAttempt
309
+
310
+ def score_final_response(content: str):
311
+ conversation = ConversationTrace(
312
+ events=(*spec.context.events, TextMessage(role="assistant", content=content))
313
+ )
314
+ return grade_answer(spec, GradingAttempt(conversation))
315
+ ```
316
+
317
+ Three functions grade an attempt:
318
+
319
+ - `taskcompendium.grading.grade_answer(task, attempt)` grades with an in-process `VerifyitGrader`. The answer format extracts the submission, or `attempt.state` supplies it for a `state` answer, and `verifyit.candidate.grade_candidate` scores it. It raises `TypeError` for other graders.
320
+ - `taskcompendium.runtime.grading.grade_in_sandbox(task, attempt, factory, machine_spec, *, task_machine=None, timeout=None)` is asynchronous. It grades with a `VerifyitGrader` that has an environment, or with a `ScriptGrader`, in a fresh machine from `factory`. `task_machine` is the agent's machine, required for `collect` and `artifacts`. Machine exceptions propagate.
321
+ - `taskcompendium.runtime.task_grading.grade_task(task, attempt, *, machine_factory=None, machine_spec=None)` grades any task synchronously. A `NoGrader` task is `unavailable`. An in-process grader uses `grade_answer`. A grader with an environment uses `grade_in_sandbox` after `grade_task` checks the factory's backend against `environment.compatible_backends`. A missing factory or machine specification, and a machine `RuntimeError` or `OSError`, become `infra_error`. A `SessionGrader` raises `TypeError`.
322
+
323
+ RolloutEngine's Shellbox session calls `grade_answer` or `grade_in_sandbox` after the turn loop; see [task rollouts](../../docs/references/task-rollouts.md).
324
+
325
+ The execution runtime completes file and environment-state acquisition before grading. `GradingAttempt` carries the conversation, captured file bytes keyed by absolute path, and an optional `StateSubmission`. A missing state is `None`; captured JSON null is `StateSubmission(None)`. `taskcompendium.runtime.models.grading_attempt(conversation, evidence)` builds an attempt from captured `RuntimeEvidence`. A grading machine also receives a conversational answer extracted from the conversation. Expected values remain in `task.grader` and verifier resources and must not be included in the model request.
326
+
327
+ A valid correct answer produces `GradeResult(status=graded, reward=1.0)`; a valid wrong answer produces `graded` with reward `0.0`. A final message the answer format cannot read, and numeric text the `numeric` mode cannot read, produce `submission_failure` with reward `0.0`. verifyit's `invalid_task` and `infra_error` statuses carry no reward and stay separate from wrong or invalid submissions. `grade_answer` raises when the answer format is incompatible with the grader.
328
+
329
+ Execution runtimes decode provider responses into `TextMessage` or `AssistantToolCalls` before constructing a `ConversationTrace`. Tool-call arguments must be decoded JSON objects. A runtime may expose malformed provider output as `submission_failure` with reward `0.0`; network and execution errors remain separate.
330
+
331
+ ## Dataset conversion
332
+
333
+ TaskSpec defines the serialized task contract. Dataset conversion pipelines own storage layout and streaming I/O, using Zephyr for Parquet processing. The optional `taskcompendium.pipeline` package stores curation outputs through Zephyr. See [pipeline contracts](src/taskcompendium/pipeline/README.md). JSON decoding preserves valid requirements independently of a runtime's support for them.
334
+
335
+ ```python
336
+ serialized = spec.model_dump_json()
337
+ restored = TaskSpec.model_validate_json(serialized)
338
+ ```
339
+
340
+ The serialized spec includes the grader configuration and verifier resources. Store it where trusted grading code can read it; build model requests with `chat_request(task)` or `render_instruction(task)`, which read only the public context, final tools, and answer format.
341
+
342
+ ## Development
343
+
344
+ TaskCompendium requires Python 3.12 or 3.13 and uses `marin-rigging` for relative POSIX path and mount-collision validation. Resource names preserve Linux semantics, including colons, backslashes, trailing spaces, and case distinctions. Absolute paths, traversal, NUL bytes, duplicate files, and file/directory collisions are rejected. The validator leaf module performs no storage access.
345
+
346
+ TaskCompendium uses the root workspace's `uv.lock` and `.venv`. Run the package tests from the repository root:
347
+
348
+ ```bash
349
+ uv run --package marin-taskcompendium --extra pipeline --group test pytest lib/taskcompendium/tests -q
350
+
351
+ # Type-check from the package project directory with its dependencies available.
352
+ cd lib/taskcompendium
353
+ uvx --from 'pyrefly>=1.0.0,<1.1.0' pyrefly check
354
+ ```
355
+
356
+ Schema `0.25` stores the answer format and the typed grader on each task. Decoders reject other schema versions; existing conversion pipelines must emit the current contract.