openpond-sdk 0.5.5 → 0.5.7
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/TRAINING_PROTOCOL.md +44 -0
- package/dist/chunk-ARUJ3FJT.js +447 -0
- package/dist/chunk-ARUJ3FJT.js.map +7 -0
- package/dist/{chunk-M6MM6IGG.js → chunk-F5JBFRAH.js} +300 -220
- package/dist/chunk-F5JBFRAH.js.map +7 -0
- package/dist/chunk-QFT2MCIC.js +1526 -0
- package/dist/chunk-QFT2MCIC.js.map +7 -0
- package/dist/{chunk-NF63YMOL.js → chunk-QUOW6URQ.js} +7 -9
- package/dist/{chunk-NF63YMOL.js.map → chunk-QUOW6URQ.js.map} +3 -3
- package/dist/{chunk-25TBEGXB.js → chunk-TKJU3BOW.js} +2 -2
- package/dist/chunk-UJZ3XEE3.js +902 -0
- package/dist/chunk-UJZ3XEE3.js.map +7 -0
- package/dist/{chunk-EV4PKJWN.js → chunk-UKVK7MRJ.js} +320 -208
- package/dist/chunk-UKVK7MRJ.js.map +7 -0
- package/dist/index.js +3 -3
- package/dist/learning.js +3 -3
- package/dist/model-taskset-runs.js +2 -2
- package/dist/refiner.js +1 -1
- package/dist/taskset-drafts.js +36 -34
- package/dist/taskset-drafts.js.map +1 -1
- package/dist/taskset-packages.js +52 -466
- package/dist/taskset-packages.js.map +4 -4
- package/dist/training-bundle.js +459 -0
- package/dist/training-bundle.js.map +7 -0
- package/dist/training.js +9 -1
- package/dist/types/packages/sdk/src/taskset-authored-contracts.d.ts +2 -2
- package/dist/types/packages/sdk/src/taskset-authored-files.d.ts +1 -1
- package/dist/types/packages/sdk/src/taskset-draft-dataset-artifacts.d.ts +2 -2
- package/dist/types/packages/sdk/src/taskset-package-projection.d.ts +1 -1
- package/dist/types/packages/sdk/src/training-bundle-contracts.d.ts +287 -0
- package/dist/types/packages/sdk/src/training-bundle-contracts.d.ts.map +1 -0
- package/dist/types/packages/sdk/src/training-bundle.d.ts +48 -0
- package/dist/types/packages/sdk/src/training-bundle.d.ts.map +1 -0
- package/dist/types/packages/sdk/src/training-candidate-decisions.d.ts +90 -0
- package/dist/types/packages/sdk/src/training-candidate-decisions.d.ts.map +1 -0
- package/dist/types/packages/sdk/src/training-learning-batch.d.ts +719 -0
- package/dist/types/packages/sdk/src/training-learning-batch.d.ts.map +1 -0
- package/dist/types/packages/sdk/src/training.d.ts +8 -0
- package/dist/types/packages/sdk/src/training.d.ts.map +1 -1
- package/package.json +6 -2
- package/dist/chunk-EV4PKJWN.js.map +0 -7
- package/dist/chunk-M6MM6IGG.js.map +0 -7
- package/dist/chunk-PXB6YY36.js +0 -2395
- package/dist/chunk-PXB6YY36.js.map +0 -7
- /package/dist/{chunk-25TBEGXB.js.map → chunk-TKJU3BOW.js.map} +0 -0
|
@@ -0,0 +1,1526 @@
|
|
|
1
|
+
import {
|
|
2
|
+
ModelStarterExecutionSchema,
|
|
3
|
+
ModelTasksetPackageSchema,
|
|
4
|
+
createModelStarterExecutionAsset,
|
|
5
|
+
modelStarterExecutionAssetId,
|
|
6
|
+
validateModelStarterExecution,
|
|
7
|
+
validateModelTasksetPackage
|
|
8
|
+
} from "./chunk-4IDTRFW2.js";
|
|
9
|
+
import {
|
|
10
|
+
ModelTasksetAuthoringSchema
|
|
11
|
+
} from "./chunk-S7I45WJQ.js";
|
|
12
|
+
import {
|
|
13
|
+
ChatModelRefSchema
|
|
14
|
+
} from "./chunk-UKVK7MRJ.js";
|
|
15
|
+
import {
|
|
16
|
+
ImmutableAssetRefSchema,
|
|
17
|
+
ImmutableReleaseRefSchema,
|
|
18
|
+
ReleaseHashSchema,
|
|
19
|
+
ReleaseIdSchema,
|
|
20
|
+
VersionedReleaseRefSchema,
|
|
21
|
+
contentHash,
|
|
22
|
+
sha256
|
|
23
|
+
} from "./chunk-QUOW6URQ.js";
|
|
24
|
+
import {
|
|
25
|
+
canonicalJson
|
|
26
|
+
} from "./chunk-ZNFOVV7B.js";
|
|
27
|
+
|
|
28
|
+
// src/taskset-draft-dataset-artifacts.ts
|
|
29
|
+
import { z } from "zod";
|
|
30
|
+
var IdSchema = z.string().trim().min(1).max(240);
|
|
31
|
+
var TimestampSchema = z.string().trim().min(1);
|
|
32
|
+
var HashSchema = z.string().trim().min(8).max(256);
|
|
33
|
+
var RelativePathSchema = z.string().trim().min(1).max(1e3).refine(
|
|
34
|
+
(value) => !value.startsWith("/") && !value.startsWith("\\") && !value.split(/[\\/]/).includes(".."),
|
|
35
|
+
"Artifact paths must be relative and stay inside the Dataset."
|
|
36
|
+
);
|
|
37
|
+
var DatasetSplitSchema = z.enum([
|
|
38
|
+
"train",
|
|
39
|
+
"validation",
|
|
40
|
+
"test",
|
|
41
|
+
"frozen_eval"
|
|
42
|
+
]);
|
|
43
|
+
var DatasetSemanticFieldSchema = z.object({
|
|
44
|
+
name: z.string().trim().min(1).max(500),
|
|
45
|
+
semanticRole: z.enum([
|
|
46
|
+
"row_id",
|
|
47
|
+
"cluster_id",
|
|
48
|
+
"prompt",
|
|
49
|
+
"messages",
|
|
50
|
+
"demonstration",
|
|
51
|
+
"chosen",
|
|
52
|
+
"rejected",
|
|
53
|
+
"expected_output",
|
|
54
|
+
"privileged_context",
|
|
55
|
+
"reward",
|
|
56
|
+
"feedback",
|
|
57
|
+
"tag",
|
|
58
|
+
"metadata"
|
|
59
|
+
]),
|
|
60
|
+
logicalType: z.enum([
|
|
61
|
+
"string",
|
|
62
|
+
"integer",
|
|
63
|
+
"float",
|
|
64
|
+
"boolean",
|
|
65
|
+
"json",
|
|
66
|
+
"messages"
|
|
67
|
+
]),
|
|
68
|
+
nullable: z.boolean(),
|
|
69
|
+
policy: z.enum(["visible", "privileged", "metadata"])
|
|
70
|
+
});
|
|
71
|
+
var DatasetSemanticSchemaSchema = z.object({
|
|
72
|
+
schemaVersion: z.literal("openpond.datasetSemanticSchema.v1"),
|
|
73
|
+
fields: z.array(DatasetSemanticFieldSchema).min(1).max(1e4),
|
|
74
|
+
schemaHash: HashSchema
|
|
75
|
+
});
|
|
76
|
+
var DatasetShardRefSchema = z.object({
|
|
77
|
+
id: IdSchema,
|
|
78
|
+
split: DatasetSplitSchema,
|
|
79
|
+
path: RelativePathSchema,
|
|
80
|
+
contentHash: HashSchema,
|
|
81
|
+
schemaHash: HashSchema,
|
|
82
|
+
sizeBytes: z.number().int().nonnegative(),
|
|
83
|
+
rowCount: z.number().int().positive(),
|
|
84
|
+
rowGroupCount: z.number().int().positive()
|
|
85
|
+
});
|
|
86
|
+
var DatasetArtifactManifestSchema = z.object({
|
|
87
|
+
schemaVersion: z.literal("openpond.datasetArtifact.v1"),
|
|
88
|
+
id: IdSchema,
|
|
89
|
+
tasksetId: IdSchema,
|
|
90
|
+
tasksetRevision: z.number().int().positive(),
|
|
91
|
+
contentHash: HashSchema,
|
|
92
|
+
format: z.literal("parquet"),
|
|
93
|
+
schema: DatasetSemanticSchemaSchema,
|
|
94
|
+
shards: z.array(DatasetShardRefSchema).min(1).max(1e5),
|
|
95
|
+
rowCount: z.number().int().positive(),
|
|
96
|
+
splitCounts: z.record(DatasetSplitSchema, z.number().int().nonnegative()),
|
|
97
|
+
sourceReceiptRefs: z.array(IdSchema).min(1).max(1e5),
|
|
98
|
+
mappingHash: HashSchema,
|
|
99
|
+
qualityReportHash: HashSchema,
|
|
100
|
+
createdAt: TimestampSchema
|
|
101
|
+
});
|
|
102
|
+
var DatasetArtifactSummarySchema = z.object({
|
|
103
|
+
schemaVersion: z.literal("openpond.datasetArtifactSummary.v1"),
|
|
104
|
+
artifactId: IdSchema,
|
|
105
|
+
tasksetId: IdSchema,
|
|
106
|
+
tasksetRevision: z.number().int().positive(),
|
|
107
|
+
format: z.literal("parquet"),
|
|
108
|
+
rowCount: z.number().int().positive(),
|
|
109
|
+
splitCounts: z.record(DatasetSplitSchema, z.number().int().nonnegative()),
|
|
110
|
+
sizeBytes: z.number().int().nonnegative(),
|
|
111
|
+
contentHash: HashSchema,
|
|
112
|
+
available: z.boolean(),
|
|
113
|
+
unavailableReason: z.string().trim().min(1).max(2e3).nullable(),
|
|
114
|
+
createdAt: TimestampSchema
|
|
115
|
+
});
|
|
116
|
+
var DatasetCatalogItemSchema = z.object({
|
|
117
|
+
schemaVersion: z.literal("openpond.datasetCatalogItem.v1"),
|
|
118
|
+
tasksetId: IdSchema,
|
|
119
|
+
tasksetRevision: z.number().int().positive(),
|
|
120
|
+
artifactId: IdSchema.nullable(),
|
|
121
|
+
name: z.string().trim().min(1).max(500),
|
|
122
|
+
status: z.enum([
|
|
123
|
+
"draft",
|
|
124
|
+
"awaiting_disclosure_approval",
|
|
125
|
+
"awaiting_materialization_approval",
|
|
126
|
+
"materializing",
|
|
127
|
+
"validating",
|
|
128
|
+
"needs_review",
|
|
129
|
+
"baselining",
|
|
130
|
+
"ready",
|
|
131
|
+
"blocked",
|
|
132
|
+
"failed",
|
|
133
|
+
"archived"
|
|
134
|
+
]),
|
|
135
|
+
storageKind: z.enum(["inline", "parquet"]),
|
|
136
|
+
rowCount: z.number().int().nonnegative(),
|
|
137
|
+
splitCounts: z.record(DatasetSplitSchema, z.number().int().nonnegative()),
|
|
138
|
+
sizeBytes: z.number().int().nonnegative().nullable(),
|
|
139
|
+
available: z.boolean(),
|
|
140
|
+
unavailableReason: z.string().trim().min(1).max(2e3).nullable(),
|
|
141
|
+
createdAt: TimestampSchema,
|
|
142
|
+
updatedAt: TimestampSchema
|
|
143
|
+
});
|
|
144
|
+
var DatasetCatalogResponseSchema = z.object({
|
|
145
|
+
schemaVersion: z.literal("openpond.datasetCatalog.v1"),
|
|
146
|
+
profileId: IdSchema,
|
|
147
|
+
datasets: z.array(DatasetCatalogItemSchema),
|
|
148
|
+
generatedAt: TimestampSchema
|
|
149
|
+
});
|
|
150
|
+
var DatasetArtifactRegistryEntrySchema = z.object({
|
|
151
|
+
schemaVersion: z.literal("openpond.datasetArtifactRegistry.v1"),
|
|
152
|
+
manifest: DatasetArtifactManifestSchema,
|
|
153
|
+
profileId: IdSchema,
|
|
154
|
+
storageRoot: z.string().trim().min(1).max(4e3),
|
|
155
|
+
relativeManifestPath: RelativePathSchema,
|
|
156
|
+
available: z.boolean(),
|
|
157
|
+
unavailableReason: z.string().trim().min(1).max(2e3).nullable(),
|
|
158
|
+
verifiedAt: TimestampSchema.nullable(),
|
|
159
|
+
createdAt: TimestampSchema,
|
|
160
|
+
updatedAt: TimestampSchema
|
|
161
|
+
});
|
|
162
|
+
var DatasetRowPageRequestSchema = z.object({
|
|
163
|
+
split: DatasetSplitSchema.nullable().default(null),
|
|
164
|
+
cursor: z.string().trim().min(1).max(2e3).nullable().default(null),
|
|
165
|
+
limit: z.number().int().min(1).max(100).default(25),
|
|
166
|
+
columns: z.array(z.string().trim().min(1).max(500)).max(100).default([])
|
|
167
|
+
});
|
|
168
|
+
var DatasetRowPageSchema = z.object({
|
|
169
|
+
schemaVersion: z.literal("openpond.datasetRowPage.v1"),
|
|
170
|
+
tasksetId: IdSchema,
|
|
171
|
+
tasksetRevision: z.number().int().positive(),
|
|
172
|
+
artifactHash: HashSchema,
|
|
173
|
+
split: DatasetSplitSchema.nullable(),
|
|
174
|
+
rows: z.array(z.record(z.string(), z.unknown())).max(100),
|
|
175
|
+
nextCursor: z.string().max(2e3).nullable(),
|
|
176
|
+
totalRows: z.number().int().nonnegative(),
|
|
177
|
+
returnedRows: z.number().int().nonnegative()
|
|
178
|
+
});
|
|
179
|
+
var DatasetStorageSettingsSchema = z.object({
|
|
180
|
+
schemaVersion: z.literal("openpond.datasetStorageSettings.v1"),
|
|
181
|
+
datasetStorePath: z.string().trim().min(1).max(4e3),
|
|
182
|
+
updatedAt: TimestampSchema
|
|
183
|
+
});
|
|
184
|
+
var DatasetStorageRootSchema = z.object({
|
|
185
|
+
id: IdSchema,
|
|
186
|
+
label: z.string().trim().min(1).max(500),
|
|
187
|
+
path: z.string().trim().min(1).max(4e3),
|
|
188
|
+
datasetStorePath: z.string().trim().min(1).max(4e3),
|
|
189
|
+
kind: z.enum(["local", "network", "removable", "cache"]),
|
|
190
|
+
configured: z.boolean(),
|
|
191
|
+
mounted: z.boolean(),
|
|
192
|
+
writable: z.boolean(),
|
|
193
|
+
totalBytes: z.number().int().nonnegative().nullable(),
|
|
194
|
+
freeBytes: z.number().int().nonnegative().nullable()
|
|
195
|
+
});
|
|
196
|
+
var DatasetStorageStateSchema = z.object({
|
|
197
|
+
schemaVersion: z.literal("openpond.datasetStorageState.v1"),
|
|
198
|
+
settings: DatasetStorageSettingsSchema,
|
|
199
|
+
storageRoots: z.array(DatasetStorageRootSchema)
|
|
200
|
+
});
|
|
201
|
+
var UpdateDatasetStorageSettingsRequestSchema = z.object({
|
|
202
|
+
datasetStorePath: z.string().trim().min(1).max(4e3)
|
|
203
|
+
});
|
|
204
|
+
|
|
205
|
+
// src/taskset-draft-dataset-sources.ts
|
|
206
|
+
import { z as z2 } from "zod";
|
|
207
|
+
var IdSchema2 = z2.string().trim().min(1).max(240);
|
|
208
|
+
var TimestampSchema2 = z2.string().trim().min(1);
|
|
209
|
+
var HashSchema2 = z2.string().trim().min(8).max(256);
|
|
210
|
+
var MetadataSchema = z2.record(z2.string(), z2.unknown()).default({});
|
|
211
|
+
var RevisionSchema = z2.string().trim().min(7).max(256);
|
|
212
|
+
var DatasetSourceReviewStatusSchema = z2.enum([
|
|
213
|
+
"pending",
|
|
214
|
+
"approved",
|
|
215
|
+
"review",
|
|
216
|
+
"blocked"
|
|
217
|
+
]);
|
|
218
|
+
var ExternalDatasetSourceBaseSchema = z2.object({
|
|
219
|
+
id: IdSchema2,
|
|
220
|
+
profileId: IdSchema2,
|
|
221
|
+
title: z2.string().trim().min(1).max(500),
|
|
222
|
+
sourceHash: HashSchema2,
|
|
223
|
+
occurredAt: TimestampSchema2,
|
|
224
|
+
licensingStatus: DatasetSourceReviewStatusSchema,
|
|
225
|
+
secretScanStatus: z2.enum(["pending", "passed", "blocked"]),
|
|
226
|
+
piiScanStatus: z2.enum(["pending", "passed", "review", "blocked"]),
|
|
227
|
+
metadata: MetadataSchema
|
|
228
|
+
});
|
|
229
|
+
var HuggingFaceDatasetSourceRefSchema = ExternalDatasetSourceBaseSchema.extend({
|
|
230
|
+
schemaVersion: z2.literal("openpond.huggingFaceDatasetSource.v1"),
|
|
231
|
+
kind: z2.literal("huggingface"),
|
|
232
|
+
repositoryId: z2.string().trim().regex(/^[^/\s]+\/[^/\s]+$/).max(500),
|
|
233
|
+
repositoryUrl: z2.string().url().max(2e3),
|
|
234
|
+
revision: RevisionSchema,
|
|
235
|
+
configuration: z2.string().trim().min(1).max(500),
|
|
236
|
+
upstreamSplits: z2.array(z2.string().trim().min(1).max(500)).min(1).max(1e3),
|
|
237
|
+
gated: z2.boolean(),
|
|
238
|
+
private: z2.boolean(),
|
|
239
|
+
declaredLicense: z2.string().trim().min(1).max(500).nullable(),
|
|
240
|
+
sourceFileHashes: z2.array(HashSchema2).min(1).max(1e5)
|
|
241
|
+
});
|
|
242
|
+
var UploadedFileDatasetSourceRefSchema = ExternalDatasetSourceBaseSchema.extend({
|
|
243
|
+
schemaVersion: z2.literal("openpond.uploadedFileDatasetSource.v1"),
|
|
244
|
+
kind: z2.literal("uploaded_file"),
|
|
245
|
+
originalFileNames: z2.array(z2.string().trim().min(1).max(500)).min(1).max(1e3),
|
|
246
|
+
mediaTypes: z2.array(z2.string().trim().min(1).max(200)).min(1).max(1e3),
|
|
247
|
+
sourceFileHashes: z2.array(HashSchema2).min(1).max(1e3),
|
|
248
|
+
totalBytes: z2.number().int().nonnegative(),
|
|
249
|
+
parserVersion: z2.string().trim().min(1).max(200)
|
|
250
|
+
});
|
|
251
|
+
var GeneratedDatasetSourceRefSchema = ExternalDatasetSourceBaseSchema.extend({
|
|
252
|
+
schemaVersion: z2.literal("openpond.generatedDatasetSource.v1"),
|
|
253
|
+
kind: z2.literal("generated"),
|
|
254
|
+
generatorId: IdSchema2,
|
|
255
|
+
generatorVersion: z2.string().trim().min(1).max(200),
|
|
256
|
+
seed: z2.number().int(),
|
|
257
|
+
generatorHash: HashSchema2
|
|
258
|
+
});
|
|
259
|
+
var LearningBatchDatasetSourceRefSchema = ExternalDatasetSourceBaseSchema.extend({
|
|
260
|
+
schemaVersion: z2.literal("openpond.learningBatchDatasetSource.v1"),
|
|
261
|
+
kind: z2.literal("learning_batch"),
|
|
262
|
+
batch: z2.object({ id: IdSchema2, revision: z2.number().int().positive(), contentHash: HashSchema2 }).strict(),
|
|
263
|
+
taskDefinition: z2.object({ id: IdSchema2, revision: z2.number().int().positive(), contentHash: HashSchema2 }).strict(),
|
|
264
|
+
admittedBy: IdSchema2
|
|
265
|
+
});
|
|
266
|
+
var ExternalDatasetSourceRefSchema = z2.discriminatedUnion("kind", [
|
|
267
|
+
HuggingFaceDatasetSourceRefSchema,
|
|
268
|
+
UploadedFileDatasetSourceRefSchema,
|
|
269
|
+
GeneratedDatasetSourceRefSchema,
|
|
270
|
+
LearningBatchDatasetSourceRefSchema
|
|
271
|
+
]);
|
|
272
|
+
|
|
273
|
+
// src/taskset-draft-harness-actions.ts
|
|
274
|
+
import { z as z3 } from "zod";
|
|
275
|
+
var HarnessActionBindingSchema = z3.object({
|
|
276
|
+
schemaVersion: z3.literal("openpond.harnessActionBinding.v1"),
|
|
277
|
+
actionId: ReleaseIdSchema,
|
|
278
|
+
modelToolName: z3.string().trim().min(1).max(64).regex(/^[a-zA-Z][a-zA-Z0-9_-]*$/),
|
|
279
|
+
description: z3.string().trim().min(1).max(1e3),
|
|
280
|
+
inputSchema: z3.record(z3.string(), z3.unknown()),
|
|
281
|
+
actionSchemaHash: ReleaseHashSchema,
|
|
282
|
+
agentRelease: ImmutableReleaseRefSchema,
|
|
283
|
+
implementationHash: ReleaseHashSchema,
|
|
284
|
+
runtimeBindingId: ReleaseIdSchema,
|
|
285
|
+
capabilityReceiptHash: ReleaseHashSchema,
|
|
286
|
+
sideEffect: z3.enum(["read", "write"]),
|
|
287
|
+
studentVisible: z3.boolean(),
|
|
288
|
+
timeoutMs: z3.number().int().positive().max(36e5),
|
|
289
|
+
episodeArgumentBindings: z3.array(
|
|
290
|
+
z3.object({
|
|
291
|
+
argument: ReleaseIdSchema,
|
|
292
|
+
source: z3.literal("case_id")
|
|
293
|
+
}).strict()
|
|
294
|
+
).max(20).default([])
|
|
295
|
+
}).strict();
|
|
296
|
+
|
|
297
|
+
// src/taskset-draft-core.ts
|
|
298
|
+
import { z as z4 } from "zod";
|
|
299
|
+
var IdSchema3 = z4.string().trim().min(1).max(240);
|
|
300
|
+
var TimestampSchema3 = z4.string().trim().min(1);
|
|
301
|
+
var HashSchema3 = z4.string().trim().min(8).max(256);
|
|
302
|
+
var Sha256Schema = z4.string().trim().regex(/^[a-f0-9]{64}$/);
|
|
303
|
+
var CodeIdentifierSchema = z4.string().trim().regex(/^[A-Za-z_$][A-Za-z0-9_$]*$/);
|
|
304
|
+
var MetadataSchema2 = z4.record(z4.string(), z4.unknown()).default({});
|
|
305
|
+
var NullableIdSchema = IdSchema3.nullable();
|
|
306
|
+
function safeRelativeFilePath(value) {
|
|
307
|
+
const normalized = value.trim().replaceAll("\\", "/");
|
|
308
|
+
if (!normalized || normalized.startsWith("/") || normalized === "." || normalized === "..") {
|
|
309
|
+
return false;
|
|
310
|
+
}
|
|
311
|
+
return !normalized.split("/").some(
|
|
312
|
+
(segment) => !segment || segment === "." || segment === ".."
|
|
313
|
+
);
|
|
314
|
+
}
|
|
315
|
+
function safeFileName(value) {
|
|
316
|
+
const normalized = value.trim();
|
|
317
|
+
return normalized.length > 0 && normalized !== "." && normalized !== ".." && !normalized.includes("/") && !normalized.includes("\\") && !normalized.includes("\0");
|
|
318
|
+
}
|
|
319
|
+
var TasksetSplitSchema = DatasetSplitSchema;
|
|
320
|
+
var TasksetPurposeSchema = z4.enum(["general", "benchmark"]);
|
|
321
|
+
var TasksetBenchmarkBindingSchema = z4.object({
|
|
322
|
+
schemaVersion: z4.literal("openpond.tasksetBenchmark.v1"),
|
|
323
|
+
definitionId: IdSchema3,
|
|
324
|
+
releaseId: IdSchema3,
|
|
325
|
+
releaseHash: Sha256Schema,
|
|
326
|
+
managedReleasePath: z4.string().trim().min(1).max(1e3).refine(safeRelativeFilePath, "Benchmark release paths must remain relative."),
|
|
327
|
+
adaptationSplit: TasksetSplitSchema,
|
|
328
|
+
evaluationSplit: TasksetSplitSchema,
|
|
329
|
+
primaryMetric: z4.enum([
|
|
330
|
+
"foreground_tokens",
|
|
331
|
+
"success_rate",
|
|
332
|
+
"latency_ms",
|
|
333
|
+
"cost_usd"
|
|
334
|
+
]),
|
|
335
|
+
qualityGate: z4.enum(["none", "non_regression", "all_pass"]),
|
|
336
|
+
source: z4.enum(["builtin", "imported"]),
|
|
337
|
+
metadata: MetadataSchema2
|
|
338
|
+
});
|
|
339
|
+
var TasksetPreferenceComparisonBindingSchema = z4.object({
|
|
340
|
+
schemaVersion: z4.literal("openpond.tasksetPreferenceComparisonBinding.v1"),
|
|
341
|
+
releaseId: IdSchema3,
|
|
342
|
+
releaseHash: Sha256Schema,
|
|
343
|
+
publishedAt: TimestampSchema3,
|
|
344
|
+
metadata: MetadataSchema2
|
|
345
|
+
});
|
|
346
|
+
var TrainingSourceConsentSchema = z4.object({
|
|
347
|
+
status: z4.enum(["pending", "granted", "denied", "revoked"]),
|
|
348
|
+
scope: z4.enum(["metadata_only", "selected_turns", "full_session"]),
|
|
349
|
+
grantedBy: NullableIdSchema,
|
|
350
|
+
grantedAt: TimestampSchema3.nullable(),
|
|
351
|
+
purpose: z4.literal("task_authoring_and_evaluation")
|
|
352
|
+
});
|
|
353
|
+
var TrainingSourceRefSchema = z4.object({
|
|
354
|
+
schemaVersion: z4.literal("openpond.trainingSource.v1"),
|
|
355
|
+
id: IdSchema3,
|
|
356
|
+
profileId: IdSchema3,
|
|
357
|
+
sessionId: IdSchema3,
|
|
358
|
+
turnIds: z4.array(IdSchema3).max(1e3).default([]),
|
|
359
|
+
workspaceId: NullableIdSchema,
|
|
360
|
+
sourceHash: HashSchema3,
|
|
361
|
+
clusterKey: IdSchema3,
|
|
362
|
+
title: z4.string().trim().min(1).max(500),
|
|
363
|
+
occurredAt: TimestampSchema3,
|
|
364
|
+
consent: TrainingSourceConsentSchema,
|
|
365
|
+
connectedAppIds: z4.array(IdSchema3).max(100).default([]),
|
|
366
|
+
secretScanStatus: z4.enum(["pending", "passed", "blocked"]),
|
|
367
|
+
piiScanStatus: z4.enum(["pending", "passed", "review", "blocked"]),
|
|
368
|
+
licensingStatus: z4.enum(["pending", "approved", "review", "blocked"]),
|
|
369
|
+
metadata: MetadataSchema2
|
|
370
|
+
});
|
|
371
|
+
var TasksetSourceRefSchema = z4.union([
|
|
372
|
+
TrainingSourceRefSchema,
|
|
373
|
+
ExternalDatasetSourceRefSchema
|
|
374
|
+
]);
|
|
375
|
+
var TaskPolicyBoundarySchema = z4.object({
|
|
376
|
+
policyVisibleFields: z4.array(IdSchema3).max(1e3).default([]),
|
|
377
|
+
privilegedFields: z4.array(IdSchema3).max(1e3).default([]),
|
|
378
|
+
hiddenGraderRefs: z4.array(IdSchema3).max(100).default([]),
|
|
379
|
+
connectedAppScopes: z4.array(IdSchema3).max(100).default([])
|
|
380
|
+
});
|
|
381
|
+
var TaskAssetRefSchema = z4.object({
|
|
382
|
+
id: IdSchema3,
|
|
383
|
+
sourceRefId: IdSchema3,
|
|
384
|
+
artifactRef: z4.string().trim().min(1).max(4e3).refine(safeRelativeFilePath, "Task asset references must be safe relative paths."),
|
|
385
|
+
fileName: z4.string().trim().min(1).max(500).refine(safeFileName, "Task asset file names must not contain path separators."),
|
|
386
|
+
mediaType: z4.string().trim().min(1).max(200),
|
|
387
|
+
sha256: Sha256Schema,
|
|
388
|
+
sizeBytes: z4.number().int().nonnegative().max(25e7),
|
|
389
|
+
split: TasksetSplitSchema,
|
|
390
|
+
metadata: MetadataSchema2
|
|
391
|
+
});
|
|
392
|
+
var TaskRequiredOutputSchema = z4.object({
|
|
393
|
+
path: z4.string().trim().min(1).max(1e3).refine(safeRelativeFilePath, "Required output paths must stay inside the Work output directory."),
|
|
394
|
+
mediaType: z4.string().trim().min(1).max(200),
|
|
395
|
+
schemaRef: IdSchema3.nullable().optional(),
|
|
396
|
+
maxBytes: z4.number().int().positive().max(1e7).optional(),
|
|
397
|
+
metadata: MetadataSchema2
|
|
398
|
+
});
|
|
399
|
+
var TaskDataRecordSchema = z4.object({
|
|
400
|
+
schemaVersion: z4.literal("openpond.taskData.v1"),
|
|
401
|
+
id: IdSchema3,
|
|
402
|
+
clusterKey: IdSchema3,
|
|
403
|
+
split: TasksetSplitSchema,
|
|
404
|
+
input: z4.record(z4.string(), z4.unknown()),
|
|
405
|
+
expectedOutput: z4.record(z4.string(), z4.unknown()).nullable(),
|
|
406
|
+
policyVisibleContext: z4.record(z4.string(), z4.unknown()).default({}),
|
|
407
|
+
privilegedContextRef: NullableIdSchema,
|
|
408
|
+
sourceRefs: z4.array(IdSchema3).min(1).max(100),
|
|
409
|
+
assets: z4.array(TaskAssetRefSchema).max(1e3).optional(),
|
|
410
|
+
resourceRefs: z4.array(IdSchema3).max(1e3).optional(),
|
|
411
|
+
requiredOutputs: z4.array(TaskRequiredOutputSchema).max(100).optional(),
|
|
412
|
+
tags: z4.array(IdSchema3).max(100).default([]),
|
|
413
|
+
metadata: MetadataSchema2
|
|
414
|
+
});
|
|
415
|
+
var TasksetEnvironmentResourceSchema = z4.object({
|
|
416
|
+
id: IdSchema3,
|
|
417
|
+
kind: z4.enum(["file", "catalog", "configuration", "code_module"]),
|
|
418
|
+
path: z4.string().trim().min(1).max(1e3).refine(safeRelativeFilePath, "Environment resource paths must remain relative."),
|
|
419
|
+
mediaType: z4.string().trim().min(1).max(200).nullable().optional(),
|
|
420
|
+
visibility: z4.enum(["policy_visible", "policy_hidden", "privileged"]),
|
|
421
|
+
required: z4.boolean(),
|
|
422
|
+
metadata: MetadataSchema2
|
|
423
|
+
});
|
|
424
|
+
var LearningSignalBaseSchema = z4.object({
|
|
425
|
+
id: IdSchema3,
|
|
426
|
+
taskId: NullableIdSchema,
|
|
427
|
+
sourceRefs: z4.array(IdSchema3).min(1).max(100),
|
|
428
|
+
artifactRef: IdSchema3,
|
|
429
|
+
approved: z4.boolean(),
|
|
430
|
+
confidence: z4.number().min(0).max(1),
|
|
431
|
+
metadata: MetadataSchema2
|
|
432
|
+
});
|
|
433
|
+
var DemonstrationSignalSchema = LearningSignalBaseSchema.extend({
|
|
434
|
+
kind: z4.literal("demonstration"),
|
|
435
|
+
prompt: z4.string().max(1e5).nullable().default(null),
|
|
436
|
+
response: z4.string().max(2e5).nullable().default(null)
|
|
437
|
+
});
|
|
438
|
+
var PreferenceSignalSchema = LearningSignalBaseSchema.extend({
|
|
439
|
+
kind: z4.literal("preference"),
|
|
440
|
+
prompt: z4.string().max(1e5),
|
|
441
|
+
chosen: z4.string().max(2e5),
|
|
442
|
+
rejected: z4.string().max(2e5),
|
|
443
|
+
rationale: z4.string().max(1e5).nullable().default(null)
|
|
444
|
+
});
|
|
445
|
+
var CorrectionSignalSchema = LearningSignalBaseSchema.extend({
|
|
446
|
+
kind: z4.literal("correction"),
|
|
447
|
+
original: z4.string().max(2e5),
|
|
448
|
+
corrected: z4.string().max(2e5),
|
|
449
|
+
rationale: z4.string().max(1e5).nullable().default(null)
|
|
450
|
+
});
|
|
451
|
+
var FeedbackSignalSchema = LearningSignalBaseSchema.extend({
|
|
452
|
+
kind: z4.literal("feedback"),
|
|
453
|
+
feedback: z4.string().max(1e5),
|
|
454
|
+
polarity: z4.enum(["positive", "negative", "mixed", "neutral"])
|
|
455
|
+
});
|
|
456
|
+
var RewardSignalSchema = LearningSignalBaseSchema.extend({
|
|
457
|
+
kind: z4.literal("reward"),
|
|
458
|
+
task: z4.string().max(1e5),
|
|
459
|
+
rules: z4.array(z4.object({
|
|
460
|
+
id: IdSchema3,
|
|
461
|
+
points: z4.number().finite(),
|
|
462
|
+
condition: z4.string().trim().min(1).max(1e5)
|
|
463
|
+
})).min(1).max(1e3),
|
|
464
|
+
otherwisePoints: z4.number().finite(),
|
|
465
|
+
executable: z4.boolean()
|
|
466
|
+
});
|
|
467
|
+
var LabelSignalSchema = LearningSignalBaseSchema.extend({
|
|
468
|
+
kind: z4.literal("label"),
|
|
469
|
+
labelKind: z4.literal("rubric"),
|
|
470
|
+
task: z4.string().max(1e5),
|
|
471
|
+
criteria: z4.array(z4.object({
|
|
472
|
+
id: IdSchema3,
|
|
473
|
+
label: z4.string().trim().min(1).max(500),
|
|
474
|
+
description: z4.string().trim().min(1).max(1e5)
|
|
475
|
+
})).min(1).max(1e3),
|
|
476
|
+
calibrationExamples: z4.object({
|
|
477
|
+
positive: z4.string().trim().min(1).max(2e5),
|
|
478
|
+
negative: z4.string().trim().min(1).max(2e5),
|
|
479
|
+
boundary: z4.string().trim().min(1).max(2e5)
|
|
480
|
+
})
|
|
481
|
+
});
|
|
482
|
+
var LearningSignalRefSchema = z4.discriminatedUnion("kind", [
|
|
483
|
+
DemonstrationSignalSchema,
|
|
484
|
+
PreferenceSignalSchema,
|
|
485
|
+
CorrectionSignalSchema,
|
|
486
|
+
FeedbackSignalSchema,
|
|
487
|
+
RewardSignalSchema,
|
|
488
|
+
LabelSignalSchema
|
|
489
|
+
]);
|
|
490
|
+
var LearningSignalInventorySchema = z4.object({
|
|
491
|
+
demonstrations: z4.array(DemonstrationSignalSchema).max(1e5).default([]),
|
|
492
|
+
preferences: z4.array(PreferenceSignalSchema).max(1e5).default([]),
|
|
493
|
+
corrections: z4.array(CorrectionSignalSchema).max(1e5).default([]),
|
|
494
|
+
feedback: z4.array(FeedbackSignalSchema).max(1e5).default([]),
|
|
495
|
+
rewards: z4.array(RewardSignalSchema).max(1e5).default([]),
|
|
496
|
+
labels: z4.array(LabelSignalSchema).max(1e5).default([])
|
|
497
|
+
});
|
|
498
|
+
var TasksetEnvironmentContractSchema = z4.object({
|
|
499
|
+
protocolVersion: z4.literal("openpond.taskEnvironment.v1"),
|
|
500
|
+
kind: z4.enum(["chat", "agent", "program", "stateful_harness", "work"]),
|
|
501
|
+
entrypoint: z4.string().trim().min(1).max(1e3),
|
|
502
|
+
stateful: z4.boolean(),
|
|
503
|
+
deterministicSeeds: z4.boolean(),
|
|
504
|
+
toolNames: z4.array(IdSchema3).max(200).default([]),
|
|
505
|
+
actionBindings: z4.array(HarnessActionBindingSchema).max(200).optional(),
|
|
506
|
+
lifecycle: z4.array(z4.enum(["create", "reset", "step", "grade", "cleanup"])).min(1),
|
|
507
|
+
defaultTimeoutMs: z4.number().int().positive().max(36e5),
|
|
508
|
+
networkPolicy: z4.enum(["none", "declared_read_only", "declared_scoped"]),
|
|
509
|
+
resources: z4.array(TasksetEnvironmentResourceSchema).max(1e4).optional(),
|
|
510
|
+
metadata: MetadataSchema2
|
|
511
|
+
});
|
|
512
|
+
var TasksetCapabilityManifestSchema = z4.object({
|
|
513
|
+
schemaVersion: z4.literal("openpond.tasksetCapabilities.v1"),
|
|
514
|
+
taskKind: z4.enum(["chat", "single_agent", "multi_agent", "custom_program"]),
|
|
515
|
+
supportedSignals: z4.array(z4.enum(["demonstration", "preference", "correction", "feedback", "reward", "label"])),
|
|
516
|
+
compatibleMethods: z4.array(z4.enum(["none", "retrieval", "sft", "dpo", "grpo", "ppo", "sdft", "opd", "opsd", "sdpo"])),
|
|
517
|
+
rewardKinds: z4.array(z4.enum(["none", "exact", "deterministic", "model_judge", "human"])),
|
|
518
|
+
requiresTools: z4.boolean(),
|
|
519
|
+
requiresState: z4.boolean(),
|
|
520
|
+
requiresPrivilegedGrading: z4.boolean(),
|
|
521
|
+
environmentPlacements: z4.array(z4.enum(["local", "remote", "colocated", "provider_native"])),
|
|
522
|
+
exportable: z4.boolean(),
|
|
523
|
+
portabilityBlockers: z4.array(z4.string().trim().min(1).max(2e3)).default([])
|
|
524
|
+
});
|
|
525
|
+
var GraderBaseSchema = z4.object({
|
|
526
|
+
id: IdSchema3,
|
|
527
|
+
version: z4.string().trim().min(1).max(100),
|
|
528
|
+
label: z4.string().trim().min(1).max(500),
|
|
529
|
+
weight: z4.number().min(0).max(1e3).default(1),
|
|
530
|
+
hardGate: z4.boolean().default(false),
|
|
531
|
+
rewardEligible: z4.boolean().default(false),
|
|
532
|
+
privileged: z4.boolean().default(false),
|
|
533
|
+
metadata: MetadataSchema2
|
|
534
|
+
});
|
|
535
|
+
var DeterministicGraderSpecSchema = GraderBaseSchema.extend({
|
|
536
|
+
kind: z4.enum(["content", "schema", "file", "diff", "test", "runtime_event", "state"]),
|
|
537
|
+
config: z4.record(z4.string(), z4.unknown())
|
|
538
|
+
});
|
|
539
|
+
var RubricGraderSpecSchema = GraderBaseSchema.extend({
|
|
540
|
+
kind: z4.literal("model_judge"),
|
|
541
|
+
rubric: z4.string().trim().min(1).max(5e4),
|
|
542
|
+
judge: ChatModelRefSchema,
|
|
543
|
+
calibrationFixtureRefs: z4.array(IdSchema3).min(1).max(500),
|
|
544
|
+
calibrationStatus: z4.enum(["pending", "passed", "failed"]),
|
|
545
|
+
temperature: z4.number().min(0).max(2).default(0)
|
|
546
|
+
});
|
|
547
|
+
var HumanGraderSpecSchema = GraderBaseSchema.extend({
|
|
548
|
+
kind: z4.literal("human"),
|
|
549
|
+
rubric: z4.string().trim().min(1).max(5e4),
|
|
550
|
+
reviewerRole: z4.string().trim().min(1).max(500)
|
|
551
|
+
});
|
|
552
|
+
var CustomVerifierGraderSpecSchema = GraderBaseSchema.extend({
|
|
553
|
+
kind: z4.literal("custom_verifier"),
|
|
554
|
+
module: z4.string().trim().min(1).max(1e3).refine(safeRelativeFilePath, "Custom verifier modules must use a safe relative path."),
|
|
555
|
+
exportName: CodeIdentifierSchema,
|
|
556
|
+
timeoutMs: z4.number().int().positive().max(3e5),
|
|
557
|
+
networkPolicy: z4.literal("none")
|
|
558
|
+
});
|
|
559
|
+
var GraderSpecSchema = z4.union([
|
|
560
|
+
DeterministicGraderSpecSchema,
|
|
561
|
+
RubricGraderSpecSchema,
|
|
562
|
+
HumanGraderSpecSchema,
|
|
563
|
+
CustomVerifierGraderSpecSchema
|
|
564
|
+
]);
|
|
565
|
+
var GraderFixtureLabelSchema = z4.enum([
|
|
566
|
+
"positive",
|
|
567
|
+
"negative",
|
|
568
|
+
"boundary",
|
|
569
|
+
"adversarial",
|
|
570
|
+
"prompt_injection",
|
|
571
|
+
"infrastructure_failure"
|
|
572
|
+
]);
|
|
573
|
+
var GraderFixtureSchema = z4.object({
|
|
574
|
+
id: IdSchema3,
|
|
575
|
+
taskId: IdSchema3,
|
|
576
|
+
label: GraderFixtureLabelSchema,
|
|
577
|
+
output: z4.record(z4.string(), z4.unknown()),
|
|
578
|
+
infrastructureError: z4.string().trim().min(1).max(1e4).nullable(),
|
|
579
|
+
expectedPassed: z4.boolean(),
|
|
580
|
+
expectedRewardEligible: z4.boolean(),
|
|
581
|
+
metadata: MetadataSchema2
|
|
582
|
+
});
|
|
583
|
+
|
|
584
|
+
// src/taskset-authored-contracts.ts
|
|
585
|
+
import { z as z5 } from "zod";
|
|
586
|
+
import { TasksetMetricPolicySchema } from "@openpond/evals/metrics";
|
|
587
|
+
var IdSchema4 = z5.string().trim().min(1).max(240);
|
|
588
|
+
var TimestampSchema4 = z5.string().trim().min(1);
|
|
589
|
+
var HashSchema4 = z5.string().trim().min(8).max(256);
|
|
590
|
+
var MetadataSchema3 = z5.record(z5.string(), z5.unknown()).default({});
|
|
591
|
+
var NullableIdSchema2 = IdSchema4.nullable();
|
|
592
|
+
var TASKSET_WORK_TOOL_NAMES = [
|
|
593
|
+
"work_capabilities",
|
|
594
|
+
"work_environment",
|
|
595
|
+
"work_list_files",
|
|
596
|
+
"work_read_file",
|
|
597
|
+
"work_read_document",
|
|
598
|
+
"work_write_file",
|
|
599
|
+
"work_write_docx",
|
|
600
|
+
"work_edit_file",
|
|
601
|
+
"work_delete_file",
|
|
602
|
+
"work_exec",
|
|
603
|
+
"work_save_output",
|
|
604
|
+
"work_stop"
|
|
605
|
+
];
|
|
606
|
+
var TasksetStatusSchema = z5.enum([
|
|
607
|
+
"draft",
|
|
608
|
+
"awaiting_disclosure_approval",
|
|
609
|
+
"awaiting_materialization_approval",
|
|
610
|
+
"materializing",
|
|
611
|
+
"validating",
|
|
612
|
+
"needs_review",
|
|
613
|
+
"baselining",
|
|
614
|
+
"ready",
|
|
615
|
+
"blocked",
|
|
616
|
+
"failed",
|
|
617
|
+
"archived"
|
|
618
|
+
]);
|
|
619
|
+
var DatasetBuildIntentSchema = z5.enum([
|
|
620
|
+
"demonstrations",
|
|
621
|
+
"preferences",
|
|
622
|
+
"verifiable_reward",
|
|
623
|
+
"rubric",
|
|
624
|
+
"discovery"
|
|
625
|
+
]);
|
|
626
|
+
var DatasetEvidenceTextSchema = z5.string().trim().max(1e5);
|
|
627
|
+
var DatasetBuildSpecificationSchema = z5.discriminatedUnion("kind", [
|
|
628
|
+
z5.object({
|
|
629
|
+
kind: z5.literal("demonstrations"),
|
|
630
|
+
behavior: DatasetEvidenceTextSchema,
|
|
631
|
+
examples: z5.array(z5.object({
|
|
632
|
+
id: IdSchema4,
|
|
633
|
+
prompt: DatasetEvidenceTextSchema,
|
|
634
|
+
response: DatasetEvidenceTextSchema
|
|
635
|
+
})).max(1e3).default([])
|
|
636
|
+
}),
|
|
637
|
+
z5.object({
|
|
638
|
+
kind: z5.literal("preferences"),
|
|
639
|
+
preference: DatasetEvidenceTextSchema,
|
|
640
|
+
pairs: z5.array(z5.object({
|
|
641
|
+
id: IdSchema4,
|
|
642
|
+
prompt: DatasetEvidenceTextSchema,
|
|
643
|
+
chosen: DatasetEvidenceTextSchema,
|
|
644
|
+
rejected: DatasetEvidenceTextSchema,
|
|
645
|
+
rationale: DatasetEvidenceTextSchema
|
|
646
|
+
})).max(1e3).default([])
|
|
647
|
+
}),
|
|
648
|
+
z5.object({
|
|
649
|
+
kind: z5.literal("verifiable_reward"),
|
|
650
|
+
task: DatasetEvidenceTextSchema,
|
|
651
|
+
rules: z5.array(z5.object({
|
|
652
|
+
id: IdSchema4,
|
|
653
|
+
points: z5.number().finite(),
|
|
654
|
+
condition: DatasetEvidenceTextSchema
|
|
655
|
+
})).max(1e3).default([]),
|
|
656
|
+
otherwisePoints: z5.number().finite().default(0)
|
|
657
|
+
}),
|
|
658
|
+
z5.object({
|
|
659
|
+
kind: z5.literal("rubric"),
|
|
660
|
+
task: DatasetEvidenceTextSchema,
|
|
661
|
+
criteria: z5.array(z5.object({
|
|
662
|
+
id: IdSchema4,
|
|
663
|
+
label: z5.string().trim().max(500),
|
|
664
|
+
description: DatasetEvidenceTextSchema
|
|
665
|
+
})).max(1e3).default([]),
|
|
666
|
+
positiveExample: DatasetEvidenceTextSchema,
|
|
667
|
+
negativeExample: DatasetEvidenceTextSchema,
|
|
668
|
+
boundaryExample: DatasetEvidenceTextSchema
|
|
669
|
+
})
|
|
670
|
+
]);
|
|
671
|
+
var GeneratedTaskFileSchema = z5.object({
|
|
672
|
+
path: z5.string().trim().min(1).max(1e3),
|
|
673
|
+
role: z5.enum(["environment", "verifier", "fixture"]),
|
|
674
|
+
content: z5.string().max(25e4)
|
|
675
|
+
});
|
|
676
|
+
var TrainingPathRecommendationSchema = z5.object({
|
|
677
|
+
primaryMethod: z5.enum(["sft", "dpo", "grpo", "ppo", "sdft", "opsd", "sdpo"]),
|
|
678
|
+
bootstrap: z5.object({
|
|
679
|
+
method: z5.literal("sft"),
|
|
680
|
+
purpose: z5.literal("trajectory_bootstrap"),
|
|
681
|
+
demonstrationRefs: z5.array(IdSchema4).min(1).max(1e5),
|
|
682
|
+
limitations: z5.array(z5.string().trim().min(1).max(5e3)).min(1).max(100)
|
|
683
|
+
}).nullable()
|
|
684
|
+
});
|
|
685
|
+
var TrainingMethodReadinessReasonCodeSchema = z5.enum([
|
|
686
|
+
"taskset_not_ready",
|
|
687
|
+
"demonstrations_missing",
|
|
688
|
+
"preference_pairs_missing",
|
|
689
|
+
"preference_pairs_invalid",
|
|
690
|
+
"executable_reward_missing",
|
|
691
|
+
"reward_not_calibrated",
|
|
692
|
+
"reward_model_missing",
|
|
693
|
+
"value_model_required",
|
|
694
|
+
"frozen_eval_missing"
|
|
695
|
+
]);
|
|
696
|
+
var TrainingMethodReadinessSchema = z5.object({
|
|
697
|
+
method: z5.enum(["sft", "dpo", "grpo", "ppo"]),
|
|
698
|
+
status: z5.enum(["recommended", "compatible", "needs_dataset_work"]),
|
|
699
|
+
reasonCodes: z5.array(TrainingMethodReadinessReasonCodeSchema).default([]),
|
|
700
|
+
reasons: z5.array(z5.string().trim().min(1).max(5e3)).default([])
|
|
701
|
+
});
|
|
702
|
+
var TasksetReadinessFindingSchema = z5.object({
|
|
703
|
+
code: IdSchema4,
|
|
704
|
+
message: z5.string().trim().min(1).max(5e3),
|
|
705
|
+
path: z5.string().trim().max(2e3).nullable()
|
|
706
|
+
});
|
|
707
|
+
var TasksetReadinessReportSchema = z5.object({
|
|
708
|
+
schemaVersion: z5.literal("openpond.tasksetReadiness.v1"),
|
|
709
|
+
tasksetId: IdSchema4,
|
|
710
|
+
tasksetHash: HashSchema4,
|
|
711
|
+
ready: z5.boolean(),
|
|
712
|
+
recommendedMethod: z5.enum(["none", "retrieval", "sft", "dpo", "grpo", "ppo", "sdft", "opd", "opsd", "sdpo"]),
|
|
713
|
+
trainingPath: TrainingPathRecommendationSchema.nullable().default(null),
|
|
714
|
+
methodReadiness: z5.array(TrainingMethodReadinessSchema).default([]),
|
|
715
|
+
compatibleDestinationClasses: z5.array(
|
|
716
|
+
z5.enum(["export", "custom", "hosted_managed"])
|
|
717
|
+
),
|
|
718
|
+
blockers: z5.array(TasksetReadinessFindingSchema).default([]),
|
|
719
|
+
advisories: z5.array(TasksetReadinessFindingSchema).default([]),
|
|
720
|
+
warnings: z5.array(z5.string().trim().min(1).max(5e3)).default([]),
|
|
721
|
+
generatedAt: TimestampSchema4
|
|
722
|
+
});
|
|
723
|
+
var AuthoringRepairSchema = z5.object({ attempt: z5.number().int().positive(), summary: z5.string().trim().min(1).max(5e3), createdAt: TimestampSchema4 });
|
|
724
|
+
var AuthoringProvenanceSchema = z5.object({
|
|
725
|
+
schemaVersion: z5.literal("openpond.taskAuthoringProvenance.v1"),
|
|
726
|
+
model: ChatModelRefSchema.nullable(),
|
|
727
|
+
modelConfig: MetadataSchema3,
|
|
728
|
+
skillHash: HashSchema4,
|
|
729
|
+
promptTemplateVersion: z5.string().trim().min(1).max(200),
|
|
730
|
+
buildIntent: DatasetBuildIntentSchema.default("demonstrations"),
|
|
731
|
+
buildSpecification: DatasetBuildSpecificationSchema.nullable().default(null),
|
|
732
|
+
evidenceHashes: z5.array(HashSchema4).max(1e5),
|
|
733
|
+
tasksetSdkVersion: z5.string().trim().min(1).max(100),
|
|
734
|
+
sourceCommit: z5.string().trim().min(1).max(256).nullable(),
|
|
735
|
+
repairHistory: z5.array(AuthoringRepairSchema).max(1e3),
|
|
736
|
+
createdAt: TimestampSchema4
|
|
737
|
+
});
|
|
738
|
+
var TasksetSchema = z5.object({
|
|
739
|
+
schemaVersion: z5.literal("openpond.taskset.v1"),
|
|
740
|
+
id: IdSchema4,
|
|
741
|
+
revision: z5.number().int().positive().default(1),
|
|
742
|
+
profileId: IdSchema4,
|
|
743
|
+
profileRelease: VersionedReleaseRefSchema.nullable().optional(),
|
|
744
|
+
createImproveRunId: NullableIdSchema2.default(null),
|
|
745
|
+
name: z5.string().trim().min(1).max(500),
|
|
746
|
+
// Portable tasks may carry all instructions in their individual inputs.
|
|
747
|
+
// Preserve an absent package prompt; draft publication validates its objective.
|
|
748
|
+
objective: z5.string().trim().max(2e4),
|
|
749
|
+
purpose: TasksetPurposeSchema.default("general"),
|
|
750
|
+
benchmark: TasksetBenchmarkBindingSchema.nullable().default(null),
|
|
751
|
+
preferenceComparison: TasksetPreferenceComparisonBindingSchema.nullable().default(null),
|
|
752
|
+
status: TasksetStatusSchema,
|
|
753
|
+
sourceRefs: z5.array(TasksetSourceRefSchema).min(1).max(1e5),
|
|
754
|
+
datasetArtifact: DatasetArtifactManifestSchema.nullable().optional(),
|
|
755
|
+
policy: TaskPolicyBoundarySchema,
|
|
756
|
+
environment: TasksetEnvironmentContractSchema,
|
|
757
|
+
capabilities: TasksetCapabilityManifestSchema,
|
|
758
|
+
metrics: TasksetMetricPolicySchema.optional(),
|
|
759
|
+
tasks: z5.array(TaskDataRecordSchema).max(1e6),
|
|
760
|
+
graders: z5.array(GraderSpecSchema).min(1).max(1e3),
|
|
761
|
+
// Imported releases are inspectable before local calibration. Admission
|
|
762
|
+
// requires real fixtures separately in validateTaskset.
|
|
763
|
+
graderFixtures: z5.array(GraderFixtureSchema).max(1e5),
|
|
764
|
+
learningSignals: LearningSignalInventorySchema,
|
|
765
|
+
authoringProvenance: AuthoringProvenanceSchema,
|
|
766
|
+
readiness: TasksetReadinessReportSchema.nullable(),
|
|
767
|
+
contentHash: HashSchema4,
|
|
768
|
+
createdAt: TimestampSchema4,
|
|
769
|
+
updatedAt: TimestampSchema4,
|
|
770
|
+
metadata: MetadataSchema3
|
|
771
|
+
}).superRefine((taskset, context) => {
|
|
772
|
+
if (taskset.purpose === "benchmark" && !taskset.benchmark) {
|
|
773
|
+
context.addIssue({
|
|
774
|
+
code: "custom",
|
|
775
|
+
message: "Benchmark Tasksets require an immutable benchmark binding.",
|
|
776
|
+
path: ["benchmark"]
|
|
777
|
+
});
|
|
778
|
+
}
|
|
779
|
+
if (taskset.purpose !== "benchmark" && taskset.benchmark) {
|
|
780
|
+
context.addIssue({
|
|
781
|
+
code: "custom",
|
|
782
|
+
message: "Only benchmark Tasksets may carry a benchmark binding.",
|
|
783
|
+
path: ["benchmark"]
|
|
784
|
+
});
|
|
785
|
+
}
|
|
786
|
+
if (taskset.datasetArtifact && taskset.tasks.length > 0) {
|
|
787
|
+
context.addIssue({
|
|
788
|
+
code: "custom",
|
|
789
|
+
message: "Artifact-backed Tasksets may not duplicate canonical rows inline.",
|
|
790
|
+
path: ["tasks"]
|
|
791
|
+
});
|
|
792
|
+
}
|
|
793
|
+
if (!taskset.datasetArtifact && taskset.tasks.length === 0) {
|
|
794
|
+
context.addIssue({
|
|
795
|
+
code: "custom",
|
|
796
|
+
message: "A Taskset requires inline tasks or a Dataset artifact manifest.",
|
|
797
|
+
path: ["tasks"]
|
|
798
|
+
});
|
|
799
|
+
}
|
|
800
|
+
});
|
|
801
|
+
function isTrainingSourceRef(source) {
|
|
802
|
+
return source.schemaVersion === "openpond.trainingSource.v1";
|
|
803
|
+
}
|
|
804
|
+
|
|
805
|
+
// src/taskset-authored-validation.ts
|
|
806
|
+
function validateTaskset(input) {
|
|
807
|
+
const parsed = TasksetSchema.safeParse(input);
|
|
808
|
+
if (!parsed.success) {
|
|
809
|
+
return {
|
|
810
|
+
valid: false,
|
|
811
|
+
taskset: null,
|
|
812
|
+
computedHash: null,
|
|
813
|
+
issues: parsed.error.issues.map((issue) => ({
|
|
814
|
+
code: "schema_invalid",
|
|
815
|
+
severity: "error",
|
|
816
|
+
message: issue.message,
|
|
817
|
+
path: issue.path.join(".")
|
|
818
|
+
}))
|
|
819
|
+
};
|
|
820
|
+
}
|
|
821
|
+
const taskset = parsed.data;
|
|
822
|
+
const issues = [];
|
|
823
|
+
validateSourceConsent(taskset, issues);
|
|
824
|
+
validateDatasetArtifact(taskset, issues);
|
|
825
|
+
validateSplitIsolation(taskset, issues);
|
|
826
|
+
validatePolicyBoundary(taskset, issues);
|
|
827
|
+
validateGraders(taskset, issues);
|
|
828
|
+
validateGraderFixtures(taskset, issues);
|
|
829
|
+
validateLearningSignals(taskset, issues);
|
|
830
|
+
validateCapabilities(taskset.capabilities, issues, taskset);
|
|
831
|
+
validateWorkExecution(taskset, issues);
|
|
832
|
+
const computedHash = tasksetContentHash(taskset);
|
|
833
|
+
if (taskset.contentHash !== computedHash) {
|
|
834
|
+
issues.push({
|
|
835
|
+
code: "content_hash_mismatch",
|
|
836
|
+
severity: "error",
|
|
837
|
+
message: `Taskset contentHash is ${taskset.contentHash}, expected ${computedHash}.`,
|
|
838
|
+
path: "contentHash"
|
|
839
|
+
});
|
|
840
|
+
}
|
|
841
|
+
return {
|
|
842
|
+
valid: !issues.some((issue) => issue.severity === "error"),
|
|
843
|
+
taskset,
|
|
844
|
+
computedHash,
|
|
845
|
+
issues
|
|
846
|
+
};
|
|
847
|
+
}
|
|
848
|
+
function validateGraderFixtures(taskset, issues) {
|
|
849
|
+
if (!taskset.graderFixtures.length) issues.push({ code: "grader_fixtures_required", severity: "error", message: "Taskset admission requires authored grader fixtures.", path: "graderFixtures" });
|
|
850
|
+
const taskIds = new Set(taskset.tasks.map((task) => task.id));
|
|
851
|
+
const required = /* @__PURE__ */ new Set(["positive", "negative", "boundary", "adversarial", "prompt_injection", "infrastructure_failure"]);
|
|
852
|
+
for (const fixture of taskset.graderFixtures) {
|
|
853
|
+
if (!taskset.datasetArtifact && !taskIds.has(fixture.taskId)) issues.push({ code: "grader_fixture_task_missing", severity: "warning", message: `Fixture ${fixture.id} references a task outside this Taskset revision (${fixture.taskId}).`, path: `graderFixtures.${fixture.id}.taskId` });
|
|
854
|
+
required.delete(fixture.label);
|
|
855
|
+
if (fixture.label === "infrastructure_failure" && !fixture.infrastructureError) issues.push({ code: "infrastructure_fixture_error_missing", severity: "error", message: `Infrastructure fixture ${fixture.id} must declare an infrastructure error.`, path: `graderFixtures.${fixture.id}.infrastructureError` });
|
|
856
|
+
}
|
|
857
|
+
for (const label of required) issues.push({ code: "grader_fixture_missing", severity: "warning", message: `Taskset does not include the optional ${label} grader calibration fixture.`, path: "graderFixtures" });
|
|
858
|
+
}
|
|
859
|
+
function validateDatasetArtifact(taskset, issues) {
|
|
860
|
+
const artifact = taskset.datasetArtifact;
|
|
861
|
+
if (!artifact) return;
|
|
862
|
+
if (artifact.tasksetId !== taskset.id || artifact.tasksetRevision !== taskset.revision) {
|
|
863
|
+
issues.push({
|
|
864
|
+
code: "dataset_artifact_taskset_mismatch",
|
|
865
|
+
severity: "error",
|
|
866
|
+
message: "Dataset artifact identity does not match its Taskset revision.",
|
|
867
|
+
path: "datasetArtifact"
|
|
868
|
+
});
|
|
869
|
+
}
|
|
870
|
+
const splitTotal = Object.values(artifact.splitCounts).reduce((total, count) => total + count, 0);
|
|
871
|
+
const shardTotal = artifact.shards.reduce((total, shard) => total + shard.rowCount, 0);
|
|
872
|
+
if (splitTotal !== artifact.rowCount || shardTotal !== artifact.rowCount) {
|
|
873
|
+
issues.push({
|
|
874
|
+
code: "dataset_artifact_row_count_mismatch",
|
|
875
|
+
severity: "error",
|
|
876
|
+
message: "Dataset artifact row, split, and shard counts do not agree.",
|
|
877
|
+
path: "datasetArtifact.rowCount"
|
|
878
|
+
});
|
|
879
|
+
}
|
|
880
|
+
const shardSplitCounts = /* @__PURE__ */ new Map();
|
|
881
|
+
for (const shard of artifact.shards) {
|
|
882
|
+
shardSplitCounts.set(
|
|
883
|
+
shard.split,
|
|
884
|
+
(shardSplitCounts.get(shard.split) ?? 0) + shard.rowCount
|
|
885
|
+
);
|
|
886
|
+
}
|
|
887
|
+
for (const [split, count] of Object.entries(artifact.splitCounts)) {
|
|
888
|
+
if ((shardSplitCounts.get(split) ?? 0) !== count) {
|
|
889
|
+
issues.push({
|
|
890
|
+
code: "dataset_artifact_split_count_mismatch",
|
|
891
|
+
severity: "error",
|
|
892
|
+
message: `Dataset artifact ${split} shard counts do not match its manifest.`,
|
|
893
|
+
path: `datasetArtifact.splitCounts.${split}`
|
|
894
|
+
});
|
|
895
|
+
}
|
|
896
|
+
}
|
|
897
|
+
const { contentHash: _contentHash, ...hashable } = artifact;
|
|
898
|
+
if (contentHash(hashable) !== artifact.contentHash) {
|
|
899
|
+
issues.push({
|
|
900
|
+
code: "dataset_artifact_hash_mismatch",
|
|
901
|
+
severity: "error",
|
|
902
|
+
message: "Dataset artifact manifest content hash is invalid.",
|
|
903
|
+
path: "datasetArtifact.contentHash"
|
|
904
|
+
});
|
|
905
|
+
}
|
|
906
|
+
}
|
|
907
|
+
function computeTasksetHash(taskset) {
|
|
908
|
+
return tasksetContentHash(taskset);
|
|
909
|
+
}
|
|
910
|
+
function tasksetContentHash(taskset) {
|
|
911
|
+
const { contentHash: _contentHash, status: _status, readiness: _readiness, updatedAt: _updatedAt, ...source } = taskset;
|
|
912
|
+
return contentHash(source);
|
|
913
|
+
}
|
|
914
|
+
function validatePortability(capabilities) {
|
|
915
|
+
const issues = [];
|
|
916
|
+
validateCapabilities(capabilities, issues, null);
|
|
917
|
+
return issues;
|
|
918
|
+
}
|
|
919
|
+
function validateSourceConsent(taskset, issues) {
|
|
920
|
+
for (const [index, source] of taskset.sourceRefs.entries()) {
|
|
921
|
+
if (isTrainingSourceRef(source) && source.consent.status !== "granted") {
|
|
922
|
+
issues.push({ code: "source_consent_missing", severity: "error", message: `Source ${source.id} is not consented.`, path: `sourceRefs.${index}.consent.status` });
|
|
923
|
+
}
|
|
924
|
+
if (source.secretScanStatus !== "passed") {
|
|
925
|
+
issues.push({ code: "source_secret_scan", severity: "error", message: `Source ${source.id} did not pass secret scanning.`, path: `sourceRefs.${index}.secretScanStatus` });
|
|
926
|
+
}
|
|
927
|
+
if (source.piiScanStatus !== "passed") {
|
|
928
|
+
issues.push({ code: "source_pii_scan", severity: "error", message: `Source ${source.id} has unresolved PII policy.`, path: `sourceRefs.${index}.piiScanStatus` });
|
|
929
|
+
}
|
|
930
|
+
if (source.licensingStatus !== "approved") {
|
|
931
|
+
issues.push({ code: "source_license", severity: "error", message: `Source ${source.id} has unresolved licensing policy.`, path: `sourceRefs.${index}.licensingStatus` });
|
|
932
|
+
}
|
|
933
|
+
}
|
|
934
|
+
}
|
|
935
|
+
function validateSplitIsolation(taskset, issues) {
|
|
936
|
+
const clusterSplits = /* @__PURE__ */ new Map();
|
|
937
|
+
for (const task of taskset.tasks) {
|
|
938
|
+
const splits = clusterSplits.get(task.clusterKey) ?? /* @__PURE__ */ new Set();
|
|
939
|
+
splits.add(task.split);
|
|
940
|
+
clusterSplits.set(task.clusterKey, splits);
|
|
941
|
+
}
|
|
942
|
+
for (const [cluster, splits] of clusterSplits) {
|
|
943
|
+
if (splits.size > 1) {
|
|
944
|
+
issues.push({ code: "split_cluster_contamination", severity: "error", message: `Source cluster ${cluster} appears in multiple splits: ${[...splits].join(", ")}.`, path: "tasks" });
|
|
945
|
+
}
|
|
946
|
+
}
|
|
947
|
+
const frozenCount = taskset.datasetArtifact ? taskset.datasetArtifact.splitCounts.frozen_eval ?? 0 : taskset.tasks.filter((task) => task.split === "frozen_eval").length;
|
|
948
|
+
if (frozenCount === 0) issues.push({ code: "frozen_eval_missing", severity: "warning", message: "Add an independent test example before training.", path: "tasks" });
|
|
949
|
+
}
|
|
950
|
+
function validateWorkExecution(taskset, issues) {
|
|
951
|
+
if (taskset.environment.kind !== "work") return;
|
|
952
|
+
if (taskset.environment.entrypoint !== "openpond-work-v1") {
|
|
953
|
+
issues.push({
|
|
954
|
+
code: "work_entrypoint_invalid",
|
|
955
|
+
severity: "error",
|
|
956
|
+
message: "Work Tasksets must use the openpond-work-v1 entrypoint.",
|
|
957
|
+
path: "environment.entrypoint"
|
|
958
|
+
});
|
|
959
|
+
}
|
|
960
|
+
if (taskset.environment.stateful) {
|
|
961
|
+
issues.push({
|
|
962
|
+
code: "work_stateful_invalid",
|
|
963
|
+
severity: "error",
|
|
964
|
+
message: "Automated Work attempts must start from clean state.",
|
|
965
|
+
path: "environment.stateful"
|
|
966
|
+
});
|
|
967
|
+
}
|
|
968
|
+
const allowedTools = new Set(TASKSET_WORK_TOOL_NAMES);
|
|
969
|
+
const unknownTools = taskset.environment.toolNames.filter(
|
|
970
|
+
(name) => !allowedTools.has(name)
|
|
971
|
+
);
|
|
972
|
+
if (unknownTools.length) {
|
|
973
|
+
issues.push({
|
|
974
|
+
code: "work_tool_unknown",
|
|
975
|
+
severity: "error",
|
|
976
|
+
message: `Work Taskset declares unknown tools: ${unknownTools.join(", ")}.`,
|
|
977
|
+
path: "environment.toolNames"
|
|
978
|
+
});
|
|
979
|
+
}
|
|
980
|
+
if (!taskset.environment.toolNames.includes("work_save_output")) {
|
|
981
|
+
issues.push({
|
|
982
|
+
code: "work_output_tool_missing",
|
|
983
|
+
severity: "error",
|
|
984
|
+
message: "Work Tasksets must declare work_save_output.",
|
|
985
|
+
path: "environment.toolNames"
|
|
986
|
+
});
|
|
987
|
+
}
|
|
988
|
+
for (const lifecycle of ["create", "reset", "step", "grade", "cleanup"]) {
|
|
989
|
+
if (!taskset.environment.lifecycle.includes(lifecycle)) {
|
|
990
|
+
issues.push({
|
|
991
|
+
code: "work_lifecycle_incomplete",
|
|
992
|
+
severity: "error",
|
|
993
|
+
message: `Work Tasksets require the ${lifecycle} lifecycle stage.`,
|
|
994
|
+
path: "environment.lifecycle"
|
|
995
|
+
});
|
|
996
|
+
}
|
|
997
|
+
}
|
|
998
|
+
const sourceById = new Map(
|
|
999
|
+
taskset.sourceRefs.map((source) => [source.id, source])
|
|
1000
|
+
);
|
|
1001
|
+
for (const [taskIndex, task] of taskset.tasks.entries()) {
|
|
1002
|
+
const assetIds = /* @__PURE__ */ new Set();
|
|
1003
|
+
const fileNames = /* @__PURE__ */ new Set();
|
|
1004
|
+
for (const [assetIndex, asset] of (task.assets ?? []).entries()) {
|
|
1005
|
+
const path = `tasks.${taskIndex}.assets.${assetIndex}`;
|
|
1006
|
+
if (assetIds.has(asset.id)) {
|
|
1007
|
+
issues.push({
|
|
1008
|
+
code: "work_asset_duplicate_id",
|
|
1009
|
+
severity: "error",
|
|
1010
|
+
message: `Task ${task.id} repeats asset id ${asset.id}.`,
|
|
1011
|
+
path: `${path}.id`
|
|
1012
|
+
});
|
|
1013
|
+
}
|
|
1014
|
+
if (fileNames.has(asset.fileName)) {
|
|
1015
|
+
issues.push({
|
|
1016
|
+
code: "work_asset_duplicate_name",
|
|
1017
|
+
severity: "error",
|
|
1018
|
+
message: `Task ${task.id} repeats staged file name ${asset.fileName}.`,
|
|
1019
|
+
path: `${path}.fileName`
|
|
1020
|
+
});
|
|
1021
|
+
}
|
|
1022
|
+
assetIds.add(asset.id);
|
|
1023
|
+
fileNames.add(asset.fileName);
|
|
1024
|
+
if (!task.sourceRefs.includes(asset.sourceRefId)) {
|
|
1025
|
+
issues.push({
|
|
1026
|
+
code: "work_asset_task_source_missing",
|
|
1027
|
+
severity: "error",
|
|
1028
|
+
message: `Asset ${asset.id} references source ${asset.sourceRefId} outside task ${task.id}.`,
|
|
1029
|
+
path: `${path}.sourceRefId`
|
|
1030
|
+
});
|
|
1031
|
+
}
|
|
1032
|
+
const source = sourceById.get(asset.sourceRefId);
|
|
1033
|
+
if (!source) {
|
|
1034
|
+
issues.push({
|
|
1035
|
+
code: "work_asset_source_missing",
|
|
1036
|
+
severity: "error",
|
|
1037
|
+
message: `Asset ${asset.id} references missing source ${asset.sourceRefId}.`,
|
|
1038
|
+
path: `${path}.sourceRefId`
|
|
1039
|
+
});
|
|
1040
|
+
} else if ("sourceFileHashes" in source && !source.sourceFileHashes.includes(asset.sha256)) {
|
|
1041
|
+
issues.push({
|
|
1042
|
+
code: "work_asset_hash_unregistered",
|
|
1043
|
+
severity: "error",
|
|
1044
|
+
message: `Asset ${asset.id} hash is not registered by source ${asset.sourceRefId}.`,
|
|
1045
|
+
path: `${path}.sha256`
|
|
1046
|
+
});
|
|
1047
|
+
}
|
|
1048
|
+
if (source && "kind" in source && source.kind === "uploaded_file" && !source.originalFileNames.includes(asset.fileName)) {
|
|
1049
|
+
issues.push({
|
|
1050
|
+
code: "work_asset_name_unregistered",
|
|
1051
|
+
severity: "error",
|
|
1052
|
+
message: `Asset ${asset.id} file name is not registered by source ${asset.sourceRefId}.`,
|
|
1053
|
+
path: `${path}.fileName`
|
|
1054
|
+
});
|
|
1055
|
+
}
|
|
1056
|
+
if (source && "kind" in source && source.kind === "uploaded_file" && !source.mediaTypes.includes(asset.mediaType)) {
|
|
1057
|
+
issues.push({
|
|
1058
|
+
code: "work_asset_media_type_unregistered",
|
|
1059
|
+
severity: "error",
|
|
1060
|
+
message: `Asset ${asset.id} media type is not registered by source ${asset.sourceRefId}.`,
|
|
1061
|
+
path: `${path}.mediaType`
|
|
1062
|
+
});
|
|
1063
|
+
}
|
|
1064
|
+
if (asset.split !== task.split) {
|
|
1065
|
+
issues.push({
|
|
1066
|
+
code: "work_asset_split_mismatch",
|
|
1067
|
+
severity: "error",
|
|
1068
|
+
message: `Asset ${asset.id} is assigned to ${asset.split}, not task split ${task.split}.`,
|
|
1069
|
+
path: `${path}.split`
|
|
1070
|
+
});
|
|
1071
|
+
}
|
|
1072
|
+
}
|
|
1073
|
+
const outputPaths = /* @__PURE__ */ new Set();
|
|
1074
|
+
for (const [outputIndex, output] of (task.requiredOutputs ?? []).entries()) {
|
|
1075
|
+
const normalized = output.path.replaceAll("\\", "/");
|
|
1076
|
+
const path = `tasks.${taskIndex}.requiredOutputs.${outputIndex}.path`;
|
|
1077
|
+
if (outputPaths.has(normalized)) {
|
|
1078
|
+
issues.push({
|
|
1079
|
+
code: "work_output_duplicate_path",
|
|
1080
|
+
severity: "error",
|
|
1081
|
+
message: `Task ${task.id} repeats required output path ${normalized}.`,
|
|
1082
|
+
path
|
|
1083
|
+
});
|
|
1084
|
+
}
|
|
1085
|
+
outputPaths.add(normalized);
|
|
1086
|
+
}
|
|
1087
|
+
if ((task.requiredOutputs ?? []).length === 0) {
|
|
1088
|
+
issues.push({
|
|
1089
|
+
code: "work_output_missing",
|
|
1090
|
+
severity: "error",
|
|
1091
|
+
message: `Work task ${task.id} must declare at least one required output.`,
|
|
1092
|
+
path: `tasks.${taskIndex}.requiredOutputs`
|
|
1093
|
+
});
|
|
1094
|
+
}
|
|
1095
|
+
}
|
|
1096
|
+
}
|
|
1097
|
+
function validatePolicyBoundary(taskset, issues) {
|
|
1098
|
+
const visible = new Set(taskset.policy.policyVisibleFields);
|
|
1099
|
+
for (const field of taskset.policy.privilegedFields) {
|
|
1100
|
+
if (visible.has(field)) issues.push({ code: "privileged_field_visible", severity: "error", message: `Field ${field} is both policy-visible and privileged.`, path: "policy" });
|
|
1101
|
+
}
|
|
1102
|
+
for (const task of taskset.tasks) {
|
|
1103
|
+
if (task.privilegedContextRef && Object.keys(task.policyVisibleContext).includes(task.privilegedContextRef)) {
|
|
1104
|
+
issues.push({ code: "privileged_context_leak", severity: "error", message: `Task ${task.id} exposes its privileged context reference.`, path: `tasks.${task.id}.policyVisibleContext` });
|
|
1105
|
+
}
|
|
1106
|
+
}
|
|
1107
|
+
}
|
|
1108
|
+
function validateGraders(taskset, issues) {
|
|
1109
|
+
const ids = /* @__PURE__ */ new Set();
|
|
1110
|
+
for (const grader of taskset.graders) {
|
|
1111
|
+
if (ids.has(grader.id)) issues.push({ code: "grader_duplicate", severity: "error", message: `Duplicate grader id ${grader.id}.`, path: "graders" });
|
|
1112
|
+
ids.add(grader.id);
|
|
1113
|
+
if (grader.kind === "model_judge" && grader.rewardEligible && grader.calibrationStatus !== "passed") {
|
|
1114
|
+
issues.push({ code: "judge_calibration_pending", severity: "warning", message: `Model judge ${grader.id} is selected for reward without a passing advisory calibration report.`, path: `graders.${grader.id}` });
|
|
1115
|
+
}
|
|
1116
|
+
if (grader.kind === "model_judge" && grader.calibrationStatus === "passed" && typeof grader.metadata.calibrationEvidenceHash !== "string") {
|
|
1117
|
+
issues.push({ code: "judge_calibration_evidence_missing", severity: "warning", message: `Model judge ${grader.id} declares calibration without an OpenPond fixture evidence hash.`, path: `graders.${grader.id}.metadata.calibrationEvidenceHash` });
|
|
1118
|
+
}
|
|
1119
|
+
if (grader.kind === "human" && grader.rewardEligible) {
|
|
1120
|
+
issues.push({ code: "human_online_reward", severity: "error", message: `Human grader ${grader.id} cannot be an online optimizer reward.`, path: `graders.${grader.id}` });
|
|
1121
|
+
}
|
|
1122
|
+
}
|
|
1123
|
+
}
|
|
1124
|
+
function validateLearningSignals(taskset, issues) {
|
|
1125
|
+
const taskIds = new Set(taskset.tasks.map((task) => task.id));
|
|
1126
|
+
const sourceIds = new Set(taskset.sourceRefs.map((source) => source.id));
|
|
1127
|
+
for (const signal of [
|
|
1128
|
+
...taskset.learningSignals.demonstrations,
|
|
1129
|
+
...taskset.learningSignals.preferences,
|
|
1130
|
+
...taskset.learningSignals.corrections,
|
|
1131
|
+
...taskset.learningSignals.feedback,
|
|
1132
|
+
...taskset.learningSignals.rewards,
|
|
1133
|
+
...taskset.learningSignals.labels
|
|
1134
|
+
]) {
|
|
1135
|
+
if (signal.taskId && !taskset.datasetArtifact && !taskIds.has(signal.taskId)) {
|
|
1136
|
+
issues.push({ code: "learning_signal_task_missing", severity: "error", message: `Signal ${signal.id} references missing task ${signal.taskId}.`, path: `learningSignals.${signal.kind}.${signal.id}.taskId` });
|
|
1137
|
+
}
|
|
1138
|
+
if (signal.sourceRefs.some((sourceId) => !sourceIds.has(sourceId))) {
|
|
1139
|
+
issues.push({ code: "learning_signal_source_missing", severity: "error", message: `Signal ${signal.id} references a source outside the Dataset.`, path: `learningSignals.${signal.kind}.${signal.id}.sourceRefs` });
|
|
1140
|
+
}
|
|
1141
|
+
}
|
|
1142
|
+
for (const preference of taskset.learningSignals.preferences) {
|
|
1143
|
+
if (preference.chosen.trim() === preference.rejected.trim()) {
|
|
1144
|
+
issues.push({ code: "preference_pair_identical", severity: "error", message: `Preference ${preference.id} has identical chosen and rejected responses.`, path: `learningSignals.preferences.${preference.id}` });
|
|
1145
|
+
}
|
|
1146
|
+
}
|
|
1147
|
+
for (const reward of taskset.learningSignals.rewards) {
|
|
1148
|
+
if (!reward.executable) {
|
|
1149
|
+
issues.push({ code: "reward_not_executable", severity: "warning", message: `Reward ${reward.id} is a reviewed specification but has no executable verifier yet.`, path: `learningSignals.rewards.${reward.id}.executable` });
|
|
1150
|
+
}
|
|
1151
|
+
}
|
|
1152
|
+
}
|
|
1153
|
+
function validateCapabilities(capabilities, issues, taskset) {
|
|
1154
|
+
if (!capabilities.exportable && capabilities.portabilityBlockers.length === 0) {
|
|
1155
|
+
issues.push({ code: "portability_reason_missing", severity: "error", message: "Non-exportable Tasksets must declare a portability blocker.", path: "capabilities.portabilityBlockers" });
|
|
1156
|
+
}
|
|
1157
|
+
if (capabilities.compatibleMethods.includes("grpo") && !capabilities.rewardKinds.some((kind) => kind === "exact" || kind === "deterministic" || kind === "model_judge")) {
|
|
1158
|
+
issues.push({ code: "grpo_reward_missing", severity: "error", message: "GRPO compatibility requires a scalar reward kind.", path: "capabilities.rewardKinds" });
|
|
1159
|
+
}
|
|
1160
|
+
if (taskset?.capabilities.compatibleMethods.includes("sft") && taskset.learningSignals.demonstrations.length === 0) {
|
|
1161
|
+
issues.push({ code: "sft_demonstrations_missing", severity: "error", message: "SFT compatibility requires approved demonstrations.", path: "learningSignals.demonstrations" });
|
|
1162
|
+
}
|
|
1163
|
+
if (taskset?.capabilities.compatibleMethods.includes("dpo") && taskset.learningSignals.preferences.length === 0) {
|
|
1164
|
+
issues.push({ code: "dpo_preferences_missing", severity: "error", message: "DPO compatibility requires chosen and rejected response pairs.", path: "learningSignals.preferences" });
|
|
1165
|
+
}
|
|
1166
|
+
if (taskset && (taskset.capabilities.compatibleMethods.includes("grpo") || taskset.capabilities.compatibleMethods.includes("ppo")) && !taskset.learningSignals.rewards.some((reward) => reward.executable)) {
|
|
1167
|
+
issues.push({ code: "online_reward_not_executable", severity: "error", message: "GRPO/PPO compatibility requires an executable scalar reward.", path: "learningSignals.rewards" });
|
|
1168
|
+
}
|
|
1169
|
+
}
|
|
1170
|
+
|
|
1171
|
+
// src/taskset-package-learning.ts
|
|
1172
|
+
import { z as z6 } from "zod";
|
|
1173
|
+
import {
|
|
1174
|
+
TaskBatchSchema,
|
|
1175
|
+
TaskEvidenceSchema,
|
|
1176
|
+
TaskAdmissionDecisionSchema,
|
|
1177
|
+
LearningSourceSchema,
|
|
1178
|
+
LearningTextAssetSchema,
|
|
1179
|
+
assertLearningContentHash,
|
|
1180
|
+
compileTaskBatch,
|
|
1181
|
+
learningRef,
|
|
1182
|
+
sameLearningRef,
|
|
1183
|
+
taskBatchPackageMetadata,
|
|
1184
|
+
verifyLearningTextAsset
|
|
1185
|
+
} from "@openpond/evals/learning";
|
|
1186
|
+
var TasksetPackageLearningResourcesSchema = z6.object({
|
|
1187
|
+
batch: TaskBatchSchema,
|
|
1188
|
+
evidence: z6.array(TaskEvidenceSchema).min(1).max(1e4),
|
|
1189
|
+
decisions: z6.array(TaskAdmissionDecisionSchema).min(1).max(1e4),
|
|
1190
|
+
sources: z6.array(LearningSourceSchema).min(1).max(1e4),
|
|
1191
|
+
assets: z6.array(LearningTextAssetSchema).max(1e3)
|
|
1192
|
+
}).strict();
|
|
1193
|
+
function learningPackageContextFiles(resources) {
|
|
1194
|
+
return resources.evidence.filter((evidence) => evidence.submission.evaluatorContext !== null).map((evidence) => {
|
|
1195
|
+
const text = canonicalJson(evidence.submission.evaluatorContext);
|
|
1196
|
+
const bytes = new TextEncoder().encode(text);
|
|
1197
|
+
return {
|
|
1198
|
+
asset: {
|
|
1199
|
+
id: `task-evidence:${evidence.contentHash}`,
|
|
1200
|
+
path: `learning/context/${evidence.contentHash}.json`,
|
|
1201
|
+
contentHash: sha256(bytes),
|
|
1202
|
+
sizeBytes: bytes.byteLength,
|
|
1203
|
+
mediaType: "application/json",
|
|
1204
|
+
visibility: "host_private"
|
|
1205
|
+
},
|
|
1206
|
+
base64: btoa(Array.from(bytes, (byte) => String.fromCharCode(byte)).join(""))
|
|
1207
|
+
};
|
|
1208
|
+
});
|
|
1209
|
+
}
|
|
1210
|
+
function validateTasksetLearningResources(taskset, input) {
|
|
1211
|
+
const resources = TasksetPackageLearningResourcesSchema.parse(input);
|
|
1212
|
+
const metadata = taskBatchPackageMetadata(taskset);
|
|
1213
|
+
if (!sameLearningRef(learningRef(resources.batch), metadata.batch)) throw new Error("Taskset learning batch differs from its admission metadata.");
|
|
1214
|
+
for (const collection of [resources.evidence, resources.decisions, resources.sources, resources.assets]) {
|
|
1215
|
+
const identities = /* @__PURE__ */ new Set();
|
|
1216
|
+
for (const resource of collection) {
|
|
1217
|
+
assertLearningContentHash(resource);
|
|
1218
|
+
const key = `${resource.id}:${resource.revision}`;
|
|
1219
|
+
if (identities.has(key)) throw new Error("Taskset learning resource identities must be unique.");
|
|
1220
|
+
identities.add(key);
|
|
1221
|
+
}
|
|
1222
|
+
}
|
|
1223
|
+
if (resources.evidence.length !== resources.batch.examples.length || resources.decisions.length !== resources.batch.examples.length) throw new Error("Taskset learning snapshots must match the admitted batch inventory.");
|
|
1224
|
+
for (const evidence of resources.evidence) {
|
|
1225
|
+
const source = resources.sources.find((source2) => sameLearningRef(learningRef(source2), evidence.source));
|
|
1226
|
+
if (!source || !sameLearningRef(source.taskDefinition, learningRef(metadata.definition))) throw new Error("Taskset evidence source differs from its task definition.");
|
|
1227
|
+
}
|
|
1228
|
+
if (resources.sources.some((source) => !resources.evidence.some((evidence) => sameLearningRef(evidence.source, learningRef(source))))) throw new Error("Taskset package includes an unrelated learning source.");
|
|
1229
|
+
const rewardAssets = metadata.rewards.flatMap((reward) => [
|
|
1230
|
+
...reward.assets,
|
|
1231
|
+
...reward.implementation.kind === "custom_verifier" ? [reward.implementation.verifierRef] : reward.implementation.kind === "model_judge" || reward.implementation.kind === "human" ? [reward.implementation.rubricRef] : [],
|
|
1232
|
+
..."inputContract" in reward.implementation ? [reward.implementation.inputContract] : []
|
|
1233
|
+
]);
|
|
1234
|
+
for (const ref of rewardAssets) {
|
|
1235
|
+
const asset = resources.assets.find((asset2) => asset2.id === ref.id);
|
|
1236
|
+
if (!asset) throw new Error("Taskset learning Reward asset is missing.");
|
|
1237
|
+
verifyLearningTextAsset(asset, ref);
|
|
1238
|
+
}
|
|
1239
|
+
if (resources.assets.some((asset) => !rewardAssets.some((ref) => ref.id === asset.id))) throw new Error("Taskset package includes an unrelated learning asset.");
|
|
1240
|
+
const compiled = compileTaskBatch({
|
|
1241
|
+
batch: resources.batch,
|
|
1242
|
+
definition: metadata.definition,
|
|
1243
|
+
binding: metadata.binding,
|
|
1244
|
+
rewards: metadata.rewards,
|
|
1245
|
+
evidence: resources.evidence,
|
|
1246
|
+
decisions: resources.decisions
|
|
1247
|
+
});
|
|
1248
|
+
const compiledMetadata = taskBatchPackageMetadata(compiled);
|
|
1249
|
+
if (contentHash(compiledMetadata) !== contentHash(metadata)) throw new Error("Taskset admission metadata differs from its reviewed snapshots.");
|
|
1250
|
+
const execution = (release) => ({ policy: release.policy, environment: release.environment, tools: release.tools });
|
|
1251
|
+
if (contentHash(execution(compiled)) !== contentHash(execution(taskset))) throw new Error("Taskset execution differs from its reviewed task definition.");
|
|
1252
|
+
const taskContent = (task) => ({
|
|
1253
|
+
id: task.id,
|
|
1254
|
+
clusterKey: task.clusterKey,
|
|
1255
|
+
split: task.split,
|
|
1256
|
+
input: task.input,
|
|
1257
|
+
expectedOutput: task.expectedOutput,
|
|
1258
|
+
policyVisibleContext: task.policyVisibleContext,
|
|
1259
|
+
privilegedContextRef: task.privilegedContextRef,
|
|
1260
|
+
artifactRefs: task.artifactRefs,
|
|
1261
|
+
requiredOutputs: task.requiredOutputs ?? [],
|
|
1262
|
+
tags: task.tags
|
|
1263
|
+
});
|
|
1264
|
+
if (contentHash(compiled.tasks.map(taskContent)) !== contentHash(taskset.tasks.map(taskContent))) throw new Error("Taskset rows differ from their reviewed evidence.");
|
|
1265
|
+
return resources;
|
|
1266
|
+
}
|
|
1267
|
+
|
|
1268
|
+
// src/taskset-package-files.ts
|
|
1269
|
+
import { z as z7 } from "zod";
|
|
1270
|
+
var MAX_TASKSET_PACKAGE_BYTES = 64 * 1024 * 1024;
|
|
1271
|
+
var TasksetPackageFileSchema = z7.object({
|
|
1272
|
+
asset: ImmutableAssetRefSchema,
|
|
1273
|
+
base64: z7.string().max(Math.ceil(MAX_TASKSET_PACKAGE_BYTES / 3) * 4)
|
|
1274
|
+
}).strict();
|
|
1275
|
+
function decodeTasksetPackageFile(value) {
|
|
1276
|
+
const file = TasksetPackageFileSchema.parse(value);
|
|
1277
|
+
const raw = atob(file.base64);
|
|
1278
|
+
if (btoa(raw) !== file.base64) throw new Error(`Taskset asset ${file.asset.id} has noncanonical base64.`);
|
|
1279
|
+
const bytes = Uint8Array.from(raw, (character) => character.charCodeAt(0));
|
|
1280
|
+
if (bytes.byteLength !== file.asset.sizeBytes || sha256(bytes) !== file.asset.contentHash) throw new Error(`Taskset asset ${file.asset.id} differs from its immutable bytes.`);
|
|
1281
|
+
return bytes;
|
|
1282
|
+
}
|
|
1283
|
+
|
|
1284
|
+
// src/taskset-package-execution.ts
|
|
1285
|
+
import { LearningTextAssetSchema as LearningTextAssetSchema2, sealLearningContent } from "@openpond/evals/learning";
|
|
1286
|
+
function resolveTasksetPackageExecution(input) {
|
|
1287
|
+
if (input.taskset.environment.entrypoint !== "openpond.javascript-environment.v1") return null;
|
|
1288
|
+
const byId = new Map(input.files.map((file) => [file.asset.id, file]));
|
|
1289
|
+
const read = (id) => {
|
|
1290
|
+
const file = byId.get(id);
|
|
1291
|
+
if (!file) throw new Error(`Taskset environment asset is missing: ${id}.`);
|
|
1292
|
+
if (file.asset.sizeBytes > 524288) throw new Error("Taskset environment source exceeds the authored asset limit.");
|
|
1293
|
+
const text = new TextDecoder("utf-8", { fatal: true }).decode(decodeTasksetPackageFile(file));
|
|
1294
|
+
return LearningTextAssetSchema2.parse(sealLearningContent({ schemaVersion: "openpond.learningTextAsset.v1", id, revision: 1, asset: file.asset, text }));
|
|
1295
|
+
};
|
|
1296
|
+
let execution;
|
|
1297
|
+
if (input.modelResources) {
|
|
1298
|
+
if (!input.modelResources.execution) throw new Error("Bound Taskset JavaScript execution resources are missing.");
|
|
1299
|
+
execution = ModelStarterExecutionSchema.parse(input.modelResources.execution);
|
|
1300
|
+
} else {
|
|
1301
|
+
const declaration = read(modelStarterExecutionAssetId(input.taskset));
|
|
1302
|
+
if (declaration.asset.visibility !== "host_private" || declaration.asset.mediaType !== "application/json" || declaration.asset.path !== "environment/execution.json") throw new Error("Taskset execution declaration must be private JSON at environment/execution.json.");
|
|
1303
|
+
execution = ModelStarterExecutionSchema.parse(JSON.parse(declaration.text));
|
|
1304
|
+
}
|
|
1305
|
+
const ids = /* @__PURE__ */ new Set([
|
|
1306
|
+
execution.javascript.module.id,
|
|
1307
|
+
...[execution.environment.actionSchemaRef, execution.environment.observationSchemaRef, execution.environment.stateSchemaRef].flatMap((ref) => ref ? [ref.id] : []),
|
|
1308
|
+
...input.taskset.tasks.flatMap((task) => task.privilegedContextRef ? [task.privilegedContextRef] : [])
|
|
1309
|
+
]);
|
|
1310
|
+
const assets = [...ids].map(read);
|
|
1311
|
+
validateModelStarterExecution(execution, input.taskset, assets);
|
|
1312
|
+
return { execution, assets };
|
|
1313
|
+
}
|
|
1314
|
+
function createTasksetPackageExecutionFile(execution) {
|
|
1315
|
+
const asset = createModelStarterExecutionAsset(execution);
|
|
1316
|
+
const bytes = new TextEncoder().encode(asset.text);
|
|
1317
|
+
let raw = "";
|
|
1318
|
+
for (const byte of bytes) raw += String.fromCharCode(byte);
|
|
1319
|
+
return { asset: asset.asset, base64: btoa(raw) };
|
|
1320
|
+
}
|
|
1321
|
+
|
|
1322
|
+
// src/taskset-package-contracts.ts
|
|
1323
|
+
import { z as z8 } from "zod";
|
|
1324
|
+
import { EnvironmentReleaseSchema, VerifierSetReleaseSchema, verifyEnvironmentRelease, verifyVerifierSetRelease } from "@openpond/evals";
|
|
1325
|
+
import { assertTasksetRelease, TasksetReleaseSchema } from "@openpond/evals/tasksets";
|
|
1326
|
+
import { assertBoundedTaskJson } from "@openpond/evals/task-schema";
|
|
1327
|
+
import { taskBatchPackageMetadata as taskBatchPackageMetadata2 } from "@openpond/evals/learning";
|
|
1328
|
+
|
|
1329
|
+
// src/model-taskset-authoring-lineage.ts
|
|
1330
|
+
import { sameLearningRef as sameLearningRef2 } from "@openpond/evals/learning";
|
|
1331
|
+
function modelAuthoredTasksetId(lineage) {
|
|
1332
|
+
return `model-authored-taskset-${contentHash({ owner: lineage.owner, root: lineage.root })}`;
|
|
1333
|
+
}
|
|
1334
|
+
function assertModelTasksetAuthoring(taskset) {
|
|
1335
|
+
if (taskset.metadata.modelTasksetAuthoring === void 0) return null;
|
|
1336
|
+
const lineage = ModelTasksetAuthoringSchema.parse(taskset.metadata.modelTasksetAuthoring);
|
|
1337
|
+
if (taskset.id !== modelAuthoredTasksetId(lineage) || (taskset.revision === 1 ? !sameLearningRef2(lineage.parent, lineage.root) : lineage.parent.id !== taskset.id || lineage.parent.revision !== taskset.revision - 1)) {
|
|
1338
|
+
throw new Error("Authored Taskset revision differs from its ownership lineage.");
|
|
1339
|
+
}
|
|
1340
|
+
return lineage;
|
|
1341
|
+
}
|
|
1342
|
+
|
|
1343
|
+
// src/taskset-package-contracts.ts
|
|
1344
|
+
var BoundModelResourcesSchema = ModelTasksetPackageSchema.omit({ taskset: true, executionResources: true });
|
|
1345
|
+
var TasksetPackageContentSchema = z8.object({
|
|
1346
|
+
schemaVersion: z8.literal("openpond.tasksetPackage.v1"),
|
|
1347
|
+
taskset: TasksetReleaseSchema,
|
|
1348
|
+
environment: EnvironmentReleaseSchema,
|
|
1349
|
+
verifierSet: VerifierSetReleaseSchema,
|
|
1350
|
+
files: z8.array(TasksetPackageFileSchema).max(1e4),
|
|
1351
|
+
modelResources: BoundModelResourcesSchema.optional(),
|
|
1352
|
+
learningResources: TasksetPackageLearningResourcesSchema.optional()
|
|
1353
|
+
}).strict();
|
|
1354
|
+
var TasksetPackageSchema = TasksetPackageContentSchema.extend({ contentHash: z8.string().regex(/^[a-f0-9]{64}$/) }).strict();
|
|
1355
|
+
function tasksetPackageRewardBinding(value) {
|
|
1356
|
+
return value.learningResources ? taskBatchPackageMetadata2(value.taskset).binding : value.modelResources?.rewardBinding ?? null;
|
|
1357
|
+
}
|
|
1358
|
+
function resolveTasksetPackageInstructions(value) {
|
|
1359
|
+
if (value.learningResources) return taskBatchPackageMetadata2(value.taskset).definition.instructions;
|
|
1360
|
+
if (value.modelResources) return value.modelResources.taskDefinition.instructions;
|
|
1361
|
+
const authoring = z8.object({ instructions: z8.string().max(2e4).optional() }).passthrough().parse(value.taskset.metadata.ordinaryAuthoring ?? {});
|
|
1362
|
+
return authoring.instructions ?? "";
|
|
1363
|
+
}
|
|
1364
|
+
function validateTasksetPackage(value) {
|
|
1365
|
+
assertBoundedTaskJson(value, MAX_TASKSET_PACKAGE_BYTES);
|
|
1366
|
+
const result = TasksetPackageSchema.parse(value);
|
|
1367
|
+
const { contentHash: hash, ...content } = result;
|
|
1368
|
+
if (contentHash(content) !== hash) throw new Error("Taskset package content hash differs from its bytes.");
|
|
1369
|
+
const { taskset, environment, verifierSet } = result;
|
|
1370
|
+
assertTasksetRelease(taskset);
|
|
1371
|
+
const authoring = assertModelTasksetAuthoring(taskset);
|
|
1372
|
+
if (authoring && (result.modelResources || result.learningResources)) throw new Error("Taskset package cannot declare competing authoring graphs.");
|
|
1373
|
+
if (!verifyEnvironmentRelease(environment) || !verifyVerifierSetRelease(verifierSet) || !same(taskset.environmentRelease, { id: environment.id, contentHash: environment.contentHash }) || !same(taskset.verifierSetRelease, { id: verifierSet.id, contentHash: verifierSet.contentHash }) || !same(taskset.environment, environment.contract) || !same(taskset.graders, verifierSet.graders)) {
|
|
1374
|
+
throw new Error("Taskset package execution releases do not match its Taskset.");
|
|
1375
|
+
}
|
|
1376
|
+
const files = /* @__PURE__ */ new Map();
|
|
1377
|
+
for (const file of result.files) {
|
|
1378
|
+
if (files.has(file.asset.id)) throw new Error(`Duplicate Taskset asset identity: ${file.asset.id}.`);
|
|
1379
|
+
decodeTasksetPackageFile(file);
|
|
1380
|
+
files.set(file.asset.id, file);
|
|
1381
|
+
}
|
|
1382
|
+
function requireAsset(ref) {
|
|
1383
|
+
const file = files.get(ref.id);
|
|
1384
|
+
if (!file || !same(file.asset, ref)) throw new Error(`Taskset package asset is missing or mismatched: ${ref.id}.`);
|
|
1385
|
+
}
|
|
1386
|
+
for (const task of taskset.tasks) {
|
|
1387
|
+
for (const asset of task.artifactRefs) requireAsset(asset);
|
|
1388
|
+
for (const output of task.requiredOutputs ?? []) if (output.schemaRef) requireAsset(output.schemaRef);
|
|
1389
|
+
if (task.privilegedContextRef && files.get(task.privilegedContextRef)?.asset.visibility !== "host_private") {
|
|
1390
|
+
throw new Error(`Taskset package private context is missing: ${task.id}.`);
|
|
1391
|
+
}
|
|
1392
|
+
}
|
|
1393
|
+
for (const ref of [environment.actionSchemaRef, environment.observationSchemaRef, environment.stateSchemaRef]) if (ref) requireAsset(ref);
|
|
1394
|
+
for (const grader of taskset.graders) {
|
|
1395
|
+
const ref = "verifierRef" in grader ? grader.verifierRef : "rubricRef" in grader ? grader.rubricRef : null;
|
|
1396
|
+
if (ref) {
|
|
1397
|
+
requireAsset(ref);
|
|
1398
|
+
if (ref.visibility === "policy") throw new Error(`Taskset grader asset must be private: ${ref.id}.`);
|
|
1399
|
+
}
|
|
1400
|
+
}
|
|
1401
|
+
if (taskset.metrics?.customAggregator) {
|
|
1402
|
+
const aggregator = taskset.metrics.customAggregator;
|
|
1403
|
+
const modules = result.files.filter((file) => file.asset.path === aggregator.module);
|
|
1404
|
+
if (modules.length !== 1 || modules[0].asset.contentHash !== aggregator.contentHash || modules[0].asset.sizeBytes > 524288 || modules[0].asset.visibility === "policy") {
|
|
1405
|
+
throw new Error("Taskset metric module is missing or mismatched, public, or exceeds the source limit.");
|
|
1406
|
+
}
|
|
1407
|
+
}
|
|
1408
|
+
for (const declarations of [taskset.metadata.environmentResources, environment.metadata.resources]) {
|
|
1409
|
+
if (declarations === void 0) continue;
|
|
1410
|
+
for (const resource of z8.array(z8.object({ path: z8.string().min(1) }).passthrough()).parse(declarations)) {
|
|
1411
|
+
if (!result.files.some((file) => file.asset.path === resource.path)) throw new Error(`Taskset environment resource is missing: ${resource.path}.`);
|
|
1412
|
+
}
|
|
1413
|
+
}
|
|
1414
|
+
if (result.learningResources) {
|
|
1415
|
+
if (result.modelResources) throw new Error("Taskset package cannot declare competing authoring graphs.");
|
|
1416
|
+
const learning = validateTasksetLearningResources(taskset, result.learningResources);
|
|
1417
|
+
for (const expected of learningPackageContextFiles(learning)) {
|
|
1418
|
+
if (!same(files.get(expected.asset.id), expected)) throw new Error("Taskset private context differs from its reviewed evidence.");
|
|
1419
|
+
}
|
|
1420
|
+
for (const asset of learning.assets) requireAsset(asset.asset);
|
|
1421
|
+
} else if (result.modelResources) {
|
|
1422
|
+
const model = validateModelTasksetPackage({ ...result.modelResources, taskset, executionResources: { environment, verifierSet } });
|
|
1423
|
+
for (const asset of model.assets) requireAsset(asset.asset);
|
|
1424
|
+
} else if (["starter", "rewardBinding", "rewardExecution", "modelTasksetDerivation", "learning"].some((key) => taskset.metadata[key] !== void 0)) {
|
|
1425
|
+
throw new Error("Bound Taskset publication requires its complete model resources.");
|
|
1426
|
+
}
|
|
1427
|
+
resolveTasksetPackageExecution(result);
|
|
1428
|
+
resolveTasksetPackageInstructions(result);
|
|
1429
|
+
return result;
|
|
1430
|
+
}
|
|
1431
|
+
function createTasksetPackage(value) {
|
|
1432
|
+
assertBoundedTaskJson(value, MAX_TASKSET_PACKAGE_BYTES);
|
|
1433
|
+
const content = TasksetPackageContentSchema.parse(value);
|
|
1434
|
+
return validateTasksetPackage({ ...content, contentHash: contentHash(content) });
|
|
1435
|
+
}
|
|
1436
|
+
function same(left, right) {
|
|
1437
|
+
return contentHash({ value: left }) === contentHash({ value: right });
|
|
1438
|
+
}
|
|
1439
|
+
|
|
1440
|
+
export {
|
|
1441
|
+
DatasetSplitSchema,
|
|
1442
|
+
DatasetSemanticFieldSchema,
|
|
1443
|
+
DatasetSemanticSchemaSchema,
|
|
1444
|
+
DatasetShardRefSchema,
|
|
1445
|
+
DatasetArtifactManifestSchema,
|
|
1446
|
+
DatasetArtifactSummarySchema,
|
|
1447
|
+
DatasetCatalogItemSchema,
|
|
1448
|
+
DatasetCatalogResponseSchema,
|
|
1449
|
+
DatasetArtifactRegistryEntrySchema,
|
|
1450
|
+
DatasetRowPageRequestSchema,
|
|
1451
|
+
DatasetRowPageSchema,
|
|
1452
|
+
DatasetStorageSettingsSchema,
|
|
1453
|
+
DatasetStorageRootSchema,
|
|
1454
|
+
DatasetStorageStateSchema,
|
|
1455
|
+
UpdateDatasetStorageSettingsRequestSchema,
|
|
1456
|
+
DatasetSourceReviewStatusSchema,
|
|
1457
|
+
HuggingFaceDatasetSourceRefSchema,
|
|
1458
|
+
UploadedFileDatasetSourceRefSchema,
|
|
1459
|
+
GeneratedDatasetSourceRefSchema,
|
|
1460
|
+
LearningBatchDatasetSourceRefSchema,
|
|
1461
|
+
ExternalDatasetSourceRefSchema,
|
|
1462
|
+
HarnessActionBindingSchema,
|
|
1463
|
+
TasksetSplitSchema,
|
|
1464
|
+
TasksetPurposeSchema,
|
|
1465
|
+
TasksetBenchmarkBindingSchema,
|
|
1466
|
+
TasksetPreferenceComparisonBindingSchema,
|
|
1467
|
+
TrainingSourceConsentSchema,
|
|
1468
|
+
TrainingSourceRefSchema,
|
|
1469
|
+
TasksetSourceRefSchema,
|
|
1470
|
+
TaskPolicyBoundarySchema,
|
|
1471
|
+
TaskAssetRefSchema,
|
|
1472
|
+
TaskRequiredOutputSchema,
|
|
1473
|
+
TaskDataRecordSchema,
|
|
1474
|
+
TasksetEnvironmentResourceSchema,
|
|
1475
|
+
DemonstrationSignalSchema,
|
|
1476
|
+
PreferenceSignalSchema,
|
|
1477
|
+
CorrectionSignalSchema,
|
|
1478
|
+
FeedbackSignalSchema,
|
|
1479
|
+
RewardSignalSchema,
|
|
1480
|
+
LabelSignalSchema,
|
|
1481
|
+
LearningSignalRefSchema,
|
|
1482
|
+
LearningSignalInventorySchema,
|
|
1483
|
+
TasksetEnvironmentContractSchema,
|
|
1484
|
+
TasksetCapabilityManifestSchema,
|
|
1485
|
+
DeterministicGraderSpecSchema,
|
|
1486
|
+
RubricGraderSpecSchema,
|
|
1487
|
+
HumanGraderSpecSchema,
|
|
1488
|
+
CustomVerifierGraderSpecSchema,
|
|
1489
|
+
GraderSpecSchema,
|
|
1490
|
+
GraderFixtureLabelSchema,
|
|
1491
|
+
GraderFixtureSchema,
|
|
1492
|
+
TASKSET_WORK_TOOL_NAMES,
|
|
1493
|
+
TasksetStatusSchema,
|
|
1494
|
+
DatasetBuildIntentSchema,
|
|
1495
|
+
DatasetBuildSpecificationSchema,
|
|
1496
|
+
GeneratedTaskFileSchema,
|
|
1497
|
+
TrainingPathRecommendationSchema,
|
|
1498
|
+
TrainingMethodReadinessReasonCodeSchema,
|
|
1499
|
+
TrainingMethodReadinessSchema,
|
|
1500
|
+
TasksetReadinessFindingSchema,
|
|
1501
|
+
TasksetReadinessReportSchema,
|
|
1502
|
+
AuthoringRepairSchema,
|
|
1503
|
+
AuthoringProvenanceSchema,
|
|
1504
|
+
TasksetSchema,
|
|
1505
|
+
isTrainingSourceRef,
|
|
1506
|
+
validateTaskset,
|
|
1507
|
+
computeTasksetHash,
|
|
1508
|
+
validatePortability,
|
|
1509
|
+
TasksetPackageLearningResourcesSchema,
|
|
1510
|
+
learningPackageContextFiles,
|
|
1511
|
+
validateTasksetLearningResources,
|
|
1512
|
+
modelAuthoredTasksetId,
|
|
1513
|
+
assertModelTasksetAuthoring,
|
|
1514
|
+
MAX_TASKSET_PACKAGE_BYTES,
|
|
1515
|
+
TasksetPackageFileSchema,
|
|
1516
|
+
decodeTasksetPackageFile,
|
|
1517
|
+
resolveTasksetPackageExecution,
|
|
1518
|
+
createTasksetPackageExecutionFile,
|
|
1519
|
+
TasksetPackageContentSchema,
|
|
1520
|
+
TasksetPackageSchema,
|
|
1521
|
+
tasksetPackageRewardBinding,
|
|
1522
|
+
resolveTasksetPackageInstructions,
|
|
1523
|
+
validateTasksetPackage,
|
|
1524
|
+
createTasksetPackage
|
|
1525
|
+
};
|
|
1526
|
+
//# sourceMappingURL=chunk-QFT2MCIC.js.map
|