openpond-sdk 0.5.7 → 0.5.9
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/TRAINING_PROTOCOL.md +20 -0
- package/dist/training-bundle.js +344 -2
- package/dist/training-bundle.js.map +4 -4
- package/dist/types/packages/sdk/src/managed-training-preparation.d.ts +201 -0
- package/dist/types/packages/sdk/src/managed-training-preparation.d.ts.map +1 -0
- package/dist/types/packages/sdk/src/model-projects.d.ts +16 -16
- package/dist/types/packages/sdk/src/taskset-draft-client.d.ts +2 -2
- package/dist/types/packages/sdk/src/taskset-package-client.d.ts +2 -2
- package/dist/types/packages/sdk/src/training-bundle.d.ts +4 -1
- package/dist/types/packages/sdk/src/training-bundle.d.ts.map +1 -1
- package/dist/types/packages/sdk/src/training-learning-batch.d.ts +709 -0
- package/dist/types/packages/sdk/src/training-learning-batch.d.ts.map +1 -1
- package/dist/types/packages/sdk/src/training-privacy.d.ts +8 -0
- package/dist/types/packages/sdk/src/training-privacy.d.ts.map +1 -0
- package/dist/types/packages/sdk/src/training-recipe-binding.d.ts +11 -0
- package/dist/types/packages/sdk/src/training-recipe-binding.d.ts.map +1 -0
- package/package.json +1 -1
package/TRAINING_PROTOCOL.md
CHANGED
|
@@ -175,3 +175,23 @@ The same entry point exports `materializeLearningBatchTaskset`. It verifies a
|
|
|
175
175
|
sealed learning batch and its evidence, admission decisions, Reward releases,
|
|
176
176
|
and verifier assets, then returns the authored Taskset, portable release, and
|
|
177
177
|
generated verifier files consumed by training preparation.
|
|
178
|
+
|
|
179
|
+
`prepareReviewedLearningBatch` applies the shared evidence privacy scan before
|
|
180
|
+
that projection. Both clients use it to construct training-source provenance
|
|
181
|
+
from the actual scan and sealed admission decisions.
|
|
182
|
+
|
|
183
|
+
`prepareManagedTrainingSubmission` constructs the staged portable artifact and
|
|
184
|
+
public policy-optimization Job from the approved manifest, recipe, exact file
|
|
185
|
+
inventory, held-out source and synchronized Model. It verifies the selected
|
|
186
|
+
bytes and budget, copies its inputs before hashing, and performs no network
|
|
187
|
+
requests. A hosted iteration supplies its durable dispatch id as
|
|
188
|
+
`idempotencyKey`; manual training retains its manifest-derived identity. The
|
|
189
|
+
caller stages the returned `artifact` and creates the returned `submission`
|
|
190
|
+
through the existing training client.
|
|
191
|
+
|
|
192
|
+
Before approval, `withAuthoritativeRecipeHashes(taskset, recipe)` binds the
|
|
193
|
+
selected Taskset and graders into the executable recipe. This is the same
|
|
194
|
+
projection used by Desktop for GRPO, PPO and DPO, including GRPO optimizer
|
|
195
|
+
semantics and bounded resource defaults. Validate the result with the applicable
|
|
196
|
+
recipe contract, then retain that exact recipe through approval and submission.
|
|
197
|
+
The shared `AdamwOptimizerConfigSchema` supplies the existing optimizer defaults.
|
package/dist/training-bundle.js
CHANGED
|
@@ -4,6 +4,7 @@ import {
|
|
|
4
4
|
resolvePortableTasksetRewardExecution
|
|
5
5
|
} from "./chunk-ARUJ3FJT.js";
|
|
6
6
|
import {
|
|
7
|
+
LearningBatchDatasetSourceRefSchema,
|
|
7
8
|
TasksetSchema,
|
|
8
9
|
computeTasksetHash
|
|
9
10
|
} from "./chunk-QFT2MCIC.js";
|
|
@@ -12,7 +13,12 @@ import "./chunk-S7I45WJQ.js";
|
|
|
12
13
|
import {
|
|
13
14
|
TRAINING_EVALUATION_SOURCE_PATH,
|
|
14
15
|
TrainingEvaluationSourceSchema,
|
|
15
|
-
assertTrainingEvaluationIsolation
|
|
16
|
+
assertTrainingEvaluationIsolation,
|
|
17
|
+
parseAndVerifyTrainingInputArtifactUpload,
|
|
18
|
+
parseAndVerifyTrainingJobSubmission,
|
|
19
|
+
trainingEvaluationSourceRef,
|
|
20
|
+
trainingInputArtifactUploadHash,
|
|
21
|
+
trainingJobSubmissionHash
|
|
16
22
|
} from "./chunk-F5JBFRAH.js";
|
|
17
23
|
import {
|
|
18
24
|
HarnessSourceSelectionSchema,
|
|
@@ -134,6 +140,42 @@ var ResolvedTrainingBundleManifestSchema = ResolvedTrainingBundleContentSchema.e
|
|
|
134
140
|
contentHash: ReleaseHashSchema
|
|
135
141
|
}).strict();
|
|
136
142
|
|
|
143
|
+
// src/training-privacy.ts
|
|
144
|
+
var SECRET_PATTERNS = [
|
|
145
|
+
{ label: "OpenAI-style API key", pattern: /\bsk-[A-Za-z0-9_-]{16,}\b/g },
|
|
146
|
+
{ label: "OpenPond API key", pattern: /\bopk_[A-Za-z0-9_-]{12,}\b/g },
|
|
147
|
+
{ label: "private key", pattern: /-----BEGIN (?:RSA |EC |OPENSSH )?PRIVATE KEY-----/g },
|
|
148
|
+
{ label: "credential assignment", pattern: /\b(?:api[_-]?key|password|secret|token)\s*[:=]\s*["']?[^\s"']{8,}/gi }
|
|
149
|
+
];
|
|
150
|
+
var EMAIL_PATTERN = /\b[A-Z0-9._%+-]+@[A-Z0-9.-]+\.[A-Z]{2,}\b/gi;
|
|
151
|
+
var PHONE_PATTERN = /\b(?:\+?1[-.\s]?)?\(?\d{3}\)?[-.\s]\d{3}[-.\s]\d{4}\b/g;
|
|
152
|
+
function scanAndRedactEvidence(text) {
|
|
153
|
+
const findings = [];
|
|
154
|
+
let redacted = text;
|
|
155
|
+
for (const item of SECRET_PATTERNS) {
|
|
156
|
+
item.pattern.lastIndex = 0;
|
|
157
|
+
if (item.pattern.test(redacted)) findings.push(item.label);
|
|
158
|
+
item.pattern.lastIndex = 0;
|
|
159
|
+
redacted = redacted.replace(item.pattern, `[REDACTED ${item.label}]`);
|
|
160
|
+
}
|
|
161
|
+
EMAIL_PATTERN.lastIndex = 0;
|
|
162
|
+
const hasEmail = EMAIL_PATTERN.test(redacted);
|
|
163
|
+
EMAIL_PATTERN.lastIndex = 0;
|
|
164
|
+
if (hasEmail) findings.push("email address");
|
|
165
|
+
redacted = redacted.replace(EMAIL_PATTERN, "[REDACTED email]");
|
|
166
|
+
PHONE_PATTERN.lastIndex = 0;
|
|
167
|
+
const hasPhone = PHONE_PATTERN.test(redacted);
|
|
168
|
+
PHONE_PATTERN.lastIndex = 0;
|
|
169
|
+
if (hasPhone) findings.push("phone number");
|
|
170
|
+
redacted = redacted.replace(PHONE_PATTERN, "[REDACTED phone]");
|
|
171
|
+
return {
|
|
172
|
+
secretStatus: findings.some((finding) => !finding.includes("email") && !finding.includes("phone")) ? "blocked" : "passed",
|
|
173
|
+
piiStatus: hasEmail || hasPhone ? "review" : "passed",
|
|
174
|
+
findings,
|
|
175
|
+
redacted
|
|
176
|
+
};
|
|
177
|
+
}
|
|
178
|
+
|
|
137
179
|
// src/training-learning-batch.ts
|
|
138
180
|
import { compileTaskBatch, learningRef, taskBatchPackageMetadata, verifyLearningTextAsset } from "@openpond/evals";
|
|
139
181
|
function materializeLearningBatchTaskset(input) {
|
|
@@ -213,6 +255,301 @@ function fixtureOutcome(grade) {
|
|
|
213
255
|
if (outcome.status !== "scored") return null;
|
|
214
256
|
return { passed: outcome.passed === true, rewardEligible: grade.training.status === "scored" };
|
|
215
257
|
}
|
|
258
|
+
function prepareReviewedLearningBatch(input) {
|
|
259
|
+
const scan = scanAndRedactEvidence(JSON.stringify({
|
|
260
|
+
tasks: input.evidence.map((item) => item.submission),
|
|
261
|
+
targets: input.decisions.map((item) => item.approvedTarget),
|
|
262
|
+
definition: input.definition
|
|
263
|
+
}));
|
|
264
|
+
if (scan.secretStatus !== "passed" || scan.piiStatus !== "passed") throw new Error(`Learning batch contains unresolved data findings (${scan.findings.join(", ")}). Correct the source examples and seal a revised batch before training.`);
|
|
265
|
+
const source = LearningBatchDatasetSourceRefSchema.parse({
|
|
266
|
+
schemaVersion: "openpond.learningBatchDatasetSource.v1",
|
|
267
|
+
kind: "learning_batch",
|
|
268
|
+
id: `batch-source-${input.batch.contentHash.slice(0, 40)}`,
|
|
269
|
+
profileId: input.profileId,
|
|
270
|
+
title: input.definition.name,
|
|
271
|
+
sourceHash: input.batch.contentHash,
|
|
272
|
+
occurredAt: input.batch.sealedAt,
|
|
273
|
+
batch: learningRef(input.batch),
|
|
274
|
+
taskDefinition: input.batch.taskDefinition,
|
|
275
|
+
admittedBy: input.batch.sealedBy,
|
|
276
|
+
licensingStatus: "approved",
|
|
277
|
+
secretScanStatus: scan.secretStatus,
|
|
278
|
+
piiScanStatus: scan.piiStatus,
|
|
279
|
+
metadata: { admission: input.admissionContext ?? "explicit_local_batch_review", privacyScanner: "openpond-evidence-v1", findings: scan.findings }
|
|
280
|
+
});
|
|
281
|
+
return materializeLearningBatchTaskset({ ...input, source });
|
|
282
|
+
}
|
|
283
|
+
|
|
284
|
+
// src/managed-training-preparation.ts
|
|
285
|
+
import { z as z2 } from "zod";
|
|
286
|
+
async function prepareManagedTrainingSubmission(raw) {
|
|
287
|
+
const input = structuredClone(raw);
|
|
288
|
+
const { project, recipe, taskset, approval } = input;
|
|
289
|
+
const manifest = HarnessRunManifestSchema.parse(input.manifest);
|
|
290
|
+
const bundle = ResolvedTrainingBundleManifestSchema.parse(input.bundleManifest);
|
|
291
|
+
const { contentHash: manifestHash, ...manifestContent } = manifest;
|
|
292
|
+
const { contentHash: bundleHash, ...bundleContent } = bundle;
|
|
293
|
+
if (contentHash(manifestContent) !== manifestHash || contentHash(bundleContent) !== bundleHash || manifest.resolvedBundleHash !== bundleHash || contentHash(manifest.harnessRelease) !== contentHash(bundle.harnessRelease) || contentHash(manifest.datasetRelease) !== contentHash(bundle.datasetRelease) || contentHash(manifest.evidenceSets) !== contentHash(bundle.evidenceSetRelease ? [bundle.evidenceSetRelease] : [])) {
|
|
294
|
+
throw new Error("The managed training manifest or resolved bundle changed.");
|
|
295
|
+
}
|
|
296
|
+
if (!project.hosted || project.hosted.syncedSourceRevision !== project.revision || project.hosted.portableProjectId !== project.id) throw new Error("Sync the exact Model Project revision before preparing managed training.");
|
|
297
|
+
const baseModel = project.trainingSetup.baseModel;
|
|
298
|
+
if (!baseModel || baseModel.modelId !== manifest.model.source || baseModel.revision !== manifest.model.revision || baseModel.tokenizerRevision !== manifest.model.tokenizerRevision || baseModel.chatTemplateHash !== manifest.model.chatTemplateHash) {
|
|
299
|
+
throw new Error("The synced Model Project base Model does not match the Run manifest.");
|
|
300
|
+
}
|
|
301
|
+
if (recipe.method !== "grpo" || contentHash(recipe) !== manifest.recipe.configHash || approval.approvalHash !== manifest.approval.approvalHash || approval.maximumSpendUsd !== manifest.approval.maximumSpendUsd) {
|
|
302
|
+
throw new Error("Managed training requires the exact approved GRPO recipe and spend ceiling.");
|
|
303
|
+
}
|
|
304
|
+
const limits = z2.object({ wallTimeMs: z2.number().positive() }).parse(recipe.resourceLimits);
|
|
305
|
+
if (input.idempotencyKey !== void 0) z2.string().trim().min(16).max(191).parse(input.idempotencyKey);
|
|
306
|
+
const files = verifyFiles(bundle, input.files);
|
|
307
|
+
const evaluationSource = TrainingEvaluationSourceSchema.parse(input.evaluationSource);
|
|
308
|
+
const evaluationFile = files.find((file) => file.path === TRAINING_EVALUATION_SOURCE_PATH);
|
|
309
|
+
if (!evaluationFile || contentHash(JSON.parse(Buffer.from(evaluationFile.content, "base64").toString("utf8"))) !== contentHash(evaluationSource)) {
|
|
310
|
+
throw new Error("The evaluation source differs from the immutable training bundle.");
|
|
311
|
+
}
|
|
312
|
+
const evaluation = await trainingEvaluationSourceRef(evaluationSource);
|
|
313
|
+
const payload = {
|
|
314
|
+
schemaVersion: "openpond.managedRlPortableSubmission.v1",
|
|
315
|
+
sourceRunRef: `openpond:model-run:${manifest.id}`,
|
|
316
|
+
name: `OpenPond Managed \xB7 ${project.id}`.slice(0, 191),
|
|
317
|
+
idempotencyKey: input.idempotencyKey ?? `openpond-managed:${manifestHash}`.slice(0, 191),
|
|
318
|
+
modelProject: { id: project.hosted.projectId, portableProjectId: project.hosted.portableProjectId },
|
|
319
|
+
manifest,
|
|
320
|
+
sourceTaskset: taskset,
|
|
321
|
+
modelImprovementQualification: input.modelImprovementQualification ?? null,
|
|
322
|
+
recipe,
|
|
323
|
+
resolvedBundle: { manifest: bundle, files },
|
|
324
|
+
validationTasks: evaluationSource.tasks,
|
|
325
|
+
validationAssets: evaluationSource.assets
|
|
326
|
+
};
|
|
327
|
+
const artifactContent = {
|
|
328
|
+
schemaVersion: "openpond.trainingInputArtifactUpload.v2",
|
|
329
|
+
kind: "portable_training_bundle",
|
|
330
|
+
idempotencyKey: `stage:${input.idempotencyKey ?? manifestHash}`,
|
|
331
|
+
sourceManifest: { id: manifest.id, contentHash: manifestHash },
|
|
332
|
+
payload
|
|
333
|
+
};
|
|
334
|
+
const artifact = await parseAndVerifyTrainingInputArtifactUpload({ ...artifactContent, contentHash: await trainingInputArtifactUploadHash(artifactContent) });
|
|
335
|
+
const jobContent = {
|
|
336
|
+
schemaVersion: "openpond.trainingJobSubmission.v2",
|
|
337
|
+
idempotencyKey: input.idempotencyKey ?? `openpond-training-v2:${manifestHash}`,
|
|
338
|
+
name: `OpenPond Managed \xB7 ${project.id}`.slice(0, 200),
|
|
339
|
+
source: {
|
|
340
|
+
modelProject: { id: project.hosted.projectId, portableProjectId: project.hosted.portableProjectId, revision: project.revision, contentHash: project.hosted.etag },
|
|
341
|
+
harnessRunManifest: artifactContent.sourceManifest,
|
|
342
|
+
harnessRelease: manifest.harnessRelease,
|
|
343
|
+
taskset,
|
|
344
|
+
tasksetRelease: { id: taskset.id, contentHash: taskset.contentHash },
|
|
345
|
+
dataset: bundle.datasetRelease,
|
|
346
|
+
evaluation,
|
|
347
|
+
evidenceSets: bundle.evidenceSetRelease ? [bundle.evidenceSetRelease] : []
|
|
348
|
+
},
|
|
349
|
+
job: { kind: "policy_optimize", baseModel, recipe, rewardSource: input.rewardSource, resumeFrom: input.resumeFrom },
|
|
350
|
+
requestedCapabilities: [
|
|
351
|
+
{ id: "managed_rl.policy.grpo", version: "1", required: true },
|
|
352
|
+
{ id: `managed_rl.rollouts.${manifest.runtimeTarget.placement}`, version: "1", required: true },
|
|
353
|
+
...project.trainingSetup.managedGpuRequirement === "h100_hbm3" ? [{ id: "managed_rl.gpu.h100_hbm3", version: "1", required: true }] : []
|
|
354
|
+
],
|
|
355
|
+
placementObjective: project.trainingSetup.managedGpuPlacementObjective,
|
|
356
|
+
budget: { maximumSpendUsd: approval.maximumSpendUsd, maximumWallSeconds: Math.ceil(limits.wallTimeMs / 1e3) },
|
|
357
|
+
approval: { ...approval, approvedAt: manifest.approval.approvedAt, exportApproved: true }
|
|
358
|
+
};
|
|
359
|
+
const submission = await parseAndVerifyTrainingJobSubmission({ ...jobContent, contentHash: await trainingJobSubmissionHash(jobContent) });
|
|
360
|
+
return { artifact, submission };
|
|
361
|
+
}
|
|
362
|
+
function verifyFiles(bundle, files) {
|
|
363
|
+
if (files.length !== bundle.files.length || new Set(files.map((file) => file.path)).size !== files.length) throw new Error("The managed bundle file inventory changed.");
|
|
364
|
+
const indexed = new Map(files.map((file) => [file.path, file]));
|
|
365
|
+
return bundle.files.map((expected) => {
|
|
366
|
+
const file = indexed.get(expected.path);
|
|
367
|
+
if (!file || file.path.includes("\\") || file.path.includes("\0") || file.path.split("/").some((part) => !part || part === "." || part === "..") || file.path === "bundle-manifest.json" || file.encoding !== "base64") throw new Error("The managed bundle contains an invalid file.");
|
|
368
|
+
const bytes2 = Buffer.from(file.content, "base64");
|
|
369
|
+
if (bytes2.toString("base64") !== file.content || file.sha256 !== expected.sha256 || file.sizeBytes !== expected.sizeBytes || bytes2.byteLength !== expected.sizeBytes || sha256(bytes2) !== expected.sha256) throw new Error(`Managed training bundle file ${file.path} changed before upload.`);
|
|
370
|
+
return file;
|
|
371
|
+
});
|
|
372
|
+
}
|
|
373
|
+
|
|
374
|
+
// src/training-recipe-binding.ts
|
|
375
|
+
import { z as z3 } from "zod";
|
|
376
|
+
var AdamwOptimizerConfigSchema = z3.object({
|
|
377
|
+
name: z3.literal("adamw").default("adamw"),
|
|
378
|
+
weightDecay: z3.number().nonnegative().max(1).default(0),
|
|
379
|
+
beta1: z3.number().positive().lt(1).default(0.9),
|
|
380
|
+
beta2: z3.number().positive().lt(1).default(0.999),
|
|
381
|
+
epsilon: z3.number().positive().max(0.01).default(1e-8)
|
|
382
|
+
}).strict().default({
|
|
383
|
+
name: "adamw",
|
|
384
|
+
weightDecay: 0,
|
|
385
|
+
beta1: 0.9,
|
|
386
|
+
beta2: 0.999,
|
|
387
|
+
epsilon: 1e-8
|
|
388
|
+
});
|
|
389
|
+
function withAuthoritativeRecipeHashes(taskset, recipe) {
|
|
390
|
+
if (!recipe || typeof recipe !== "object" || Array.isArray(recipe)) {
|
|
391
|
+
return recipe;
|
|
392
|
+
}
|
|
393
|
+
const candidate = recipe;
|
|
394
|
+
if (candidate.method === "dpo") {
|
|
395
|
+
const policyModel = record(candidate.policyModel);
|
|
396
|
+
const referenceModel = record(candidate.referenceModel);
|
|
397
|
+
const dataset2 = record(candidate.dataset);
|
|
398
|
+
const invalidationHash = contentHash({
|
|
399
|
+
tasksetHash: taskset.contentHash,
|
|
400
|
+
preferenceSignals: taskset.learningSignals.preferences.map((signal) => ({
|
|
401
|
+
id: signal.id,
|
|
402
|
+
artifactRef: signal.artifactRef,
|
|
403
|
+
prompt: signal.prompt,
|
|
404
|
+
chosen: signal.chosen,
|
|
405
|
+
rejected: signal.rejected,
|
|
406
|
+
approved: signal.approved
|
|
407
|
+
})),
|
|
408
|
+
policyModel,
|
|
409
|
+
referenceModel,
|
|
410
|
+
dataset: dataset2
|
|
411
|
+
});
|
|
412
|
+
return {
|
|
413
|
+
...candidate,
|
|
414
|
+
referenceLogprobs: {
|
|
415
|
+
cacheSchemaVersion: "openpond.dpoReferenceLogprobs.v1",
|
|
416
|
+
cacheKey: contentHash(["dpo-reference-logprobs", invalidationHash]),
|
|
417
|
+
invalidationHash
|
|
418
|
+
}
|
|
419
|
+
};
|
|
420
|
+
}
|
|
421
|
+
if (candidate.method === "ppo") {
|
|
422
|
+
const policyOptimization = record(candidate.policyOptimization);
|
|
423
|
+
const policyModel = record(policyOptimization.policyModel);
|
|
424
|
+
const referenceModel = record(policyOptimization.referenceModel);
|
|
425
|
+
const optimizer2 = record(policyOptimization.optimizer);
|
|
426
|
+
const valueModel = record(optimizer2.valueModel);
|
|
427
|
+
const dataset2 = record(policyOptimization.dataset);
|
|
428
|
+
const reward2 = record(policyOptimization.reward);
|
|
429
|
+
const policyHash = contentHash(policyModel);
|
|
430
|
+
const referenceHash = contentHash(referenceModel);
|
|
431
|
+
const valueModelHash = contentHash(valueModel);
|
|
432
|
+
return {
|
|
433
|
+
...candidate,
|
|
434
|
+
policyOptimization: {
|
|
435
|
+
...policyOptimization,
|
|
436
|
+
dataset: {
|
|
437
|
+
...dataset2,
|
|
438
|
+
tasksetId: taskset.id,
|
|
439
|
+
tasksetHash: taskset.contentHash
|
|
440
|
+
},
|
|
441
|
+
reward: {
|
|
442
|
+
...reward2,
|
|
443
|
+
graderHash: contentHash(taskset.graders)
|
|
444
|
+
}
|
|
445
|
+
},
|
|
446
|
+
resume: {
|
|
447
|
+
...record(candidate.resume),
|
|
448
|
+
policyHash,
|
|
449
|
+
referenceHash,
|
|
450
|
+
valueModelHash
|
|
451
|
+
}
|
|
452
|
+
};
|
|
453
|
+
}
|
|
454
|
+
if (candidate.method !== "grpo") return recipe;
|
|
455
|
+
const reward = candidate.reward && typeof candidate.reward === "object" && !Array.isArray(candidate.reward) ? candidate.reward : {};
|
|
456
|
+
const metadataToolContractHash = taskset.environment.metadata.toolContractHash;
|
|
457
|
+
const authoritativeToolContractHash = typeof metadataToolContractHash === "string" && metadataToolContractHash.trim() ? metadataToolContractHash : reward.toolContractHash;
|
|
458
|
+
const baseModel = record(candidate.baseModel);
|
|
459
|
+
const dataset = record(candidate.dataset);
|
|
460
|
+
const rollout = record(candidate.rollout);
|
|
461
|
+
const optimizer = record(candidate.optimizer);
|
|
462
|
+
const loss = record(candidate.loss);
|
|
463
|
+
const resourceLimits = record(candidate.resourceLimits);
|
|
464
|
+
const maxExamples = positiveInteger(dataset.maxExamples, 1);
|
|
465
|
+
const groupSize = positiveInteger(rollout.groupSize, 2);
|
|
466
|
+
const maxOutputTokens = positiveInteger(rollout.maxOutputTokens, 1);
|
|
467
|
+
const maxPromptTokens = positiveInteger(dataset.maxPromptTokens, 1);
|
|
468
|
+
const maxSteps = positiveInteger(optimizer.maxSteps, 1);
|
|
469
|
+
const optimizerIterations = positiveInteger(optimizer.iterations, 2);
|
|
470
|
+
return {
|
|
471
|
+
...candidate,
|
|
472
|
+
resourceLimits: {
|
|
473
|
+
...resourceLimits,
|
|
474
|
+
maxGpuSeconds: resourceLimits.maxGpuSeconds ?? Math.min(
|
|
475
|
+
10800,
|
|
476
|
+
Math.ceil(positiveInteger(resourceLimits.wallTimeMs, 18e4) / 1e3)
|
|
477
|
+
)
|
|
478
|
+
},
|
|
479
|
+
reward: {
|
|
480
|
+
...reward,
|
|
481
|
+
graderHash: contentHash(taskset.graders),
|
|
482
|
+
toolContractHash: authoritativeToolContractHash
|
|
483
|
+
},
|
|
484
|
+
policyOptimization: {
|
|
485
|
+
schemaVersion: "openpond.policyOptimization.v1",
|
|
486
|
+
policyModel: baseModel,
|
|
487
|
+
referenceModel: baseModel,
|
|
488
|
+
dataset: {
|
|
489
|
+
tasksetId: taskset.id,
|
|
490
|
+
tasksetHash: taskset.contentHash,
|
|
491
|
+
split: "train",
|
|
492
|
+
selectionStrategy: dataset.selectionStrategy,
|
|
493
|
+
selectionSeed: rollout.seed,
|
|
494
|
+
maxExamples
|
|
495
|
+
},
|
|
496
|
+
sampler: {
|
|
497
|
+
temperature: rollout.temperature,
|
|
498
|
+
topP: rollout.topP,
|
|
499
|
+
maxOutputTokens,
|
|
500
|
+
maxTurns: rollout.maxTurns,
|
|
501
|
+
concurrency: rollout.concurrency
|
|
502
|
+
},
|
|
503
|
+
environment: {
|
|
504
|
+
id: reward.environmentId,
|
|
505
|
+
version: reward.environmentVersion,
|
|
506
|
+
toolContractHash: authoritativeToolContractHash
|
|
507
|
+
},
|
|
508
|
+
reward: {
|
|
509
|
+
graderId: reward.graderId,
|
|
510
|
+
graderHash: contentHash(taskset.graders),
|
|
511
|
+
learnedPreference: reward.learnedPreference ?? null
|
|
512
|
+
},
|
|
513
|
+
kl: {
|
|
514
|
+
coefficient: loss.klBeta ?? null,
|
|
515
|
+
referenceConstraint: "fixed_reference"
|
|
516
|
+
},
|
|
517
|
+
budgets: {
|
|
518
|
+
maxRollouts: positiveInteger(resourceLimits.maxRollouts, maxExamples * groupSize),
|
|
519
|
+
maxEnvironmentExecutions: positiveInteger(resourceLimits.maxRollouts, maxExamples * groupSize),
|
|
520
|
+
maxInputTokens: maxExamples * groupSize * maxPromptTokens,
|
|
521
|
+
maxOutputTokens: maxExamples * groupSize * maxOutputTokens,
|
|
522
|
+
maxOptimizerSteps: maxSteps * optimizerIterations,
|
|
523
|
+
wallTimeMs: positiveInteger(resourceLimits.wallTimeMs, 18e4),
|
|
524
|
+
maximumCostUsd: null
|
|
525
|
+
},
|
|
526
|
+
checkpointEverySteps: 1,
|
|
527
|
+
seed: rollout.seed,
|
|
528
|
+
evaluationSplit: "frozen_eval",
|
|
529
|
+
optimizer: {
|
|
530
|
+
method: "grpo",
|
|
531
|
+
groupSize,
|
|
532
|
+
normalization: "group_standardized",
|
|
533
|
+
advantageEpsilon: typeof optimizer.advantageEpsilon === "number" ? optimizer.advantageEpsilon : 1e-8,
|
|
534
|
+
loss: loss.method ?? "grpo",
|
|
535
|
+
clipRange: typeof optimizer.clipRange === "number" ? optimizer.clipRange : 0.2,
|
|
536
|
+
iterations: optimizerIterations,
|
|
537
|
+
microbatchSize: positiveInteger(optimizer.microbatchSize, 1),
|
|
538
|
+
gradientAccumulationSteps: positiveInteger(
|
|
539
|
+
optimizer.gradientAccumulationSteps,
|
|
540
|
+
1
|
|
541
|
+
),
|
|
542
|
+
adamw: AdamwOptimizerConfigSchema.parse(optimizer.adamw)
|
|
543
|
+
}
|
|
544
|
+
}
|
|
545
|
+
};
|
|
546
|
+
}
|
|
547
|
+
function record(value) {
|
|
548
|
+
return value && typeof value === "object" && !Array.isArray(value) ? value : {};
|
|
549
|
+
}
|
|
550
|
+
function positiveInteger(value, fallback) {
|
|
551
|
+
return typeof value === "number" && Number.isInteger(value) && value > 0 ? value : fallback;
|
|
552
|
+
}
|
|
216
553
|
|
|
217
554
|
// src/training-bundle.ts
|
|
218
555
|
function buildTasksetTrainingBundle(input) {
|
|
@@ -444,6 +781,7 @@ function bytes(value) {
|
|
|
444
781
|
return Buffer.from(canonicalJson(value), "utf8");
|
|
445
782
|
}
|
|
446
783
|
export {
|
|
784
|
+
AdamwOptimizerConfigSchema,
|
|
447
785
|
ComputeTargetBindingSchema,
|
|
448
786
|
HarnessRunManifestContentSchema,
|
|
449
787
|
HarnessRunManifestSchema,
|
|
@@ -454,6 +792,10 @@ export {
|
|
|
454
792
|
ScopedSecretDeclarationSchema,
|
|
455
793
|
TrainingEngineBindingSchema,
|
|
456
794
|
buildTasksetTrainingBundle,
|
|
457
|
-
materializeLearningBatchTaskset
|
|
795
|
+
materializeLearningBatchTaskset,
|
|
796
|
+
prepareManagedTrainingSubmission,
|
|
797
|
+
prepareReviewedLearningBatch,
|
|
798
|
+
scanAndRedactEvidence,
|
|
799
|
+
withAuthoritativeRecipeHashes
|
|
458
800
|
};
|
|
459
801
|
//# sourceMappingURL=training-bundle.js.map
|