openpond-sdk 0.5.8 → 0.5.10

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -2,12 +2,12 @@ import {
2
2
  learningVerifierModule,
3
3
  projectLearningBatchGraders,
4
4
  resolvePortableTasksetRewardExecution
5
- } from "./chunk-ARUJ3FJT.js";
5
+ } from "./chunk-7VZB24CH.js";
6
6
  import {
7
7
  LearningBatchDatasetSourceRefSchema,
8
8
  TasksetSchema,
9
9
  computeTasksetHash
10
- } from "./chunk-QFT2MCIC.js";
10
+ } from "./chunk-WSTMQGOY.js";
11
11
  import "./chunk-4IDTRFW2.js";
12
12
  import "./chunk-S7I45WJQ.js";
13
13
  import {
@@ -22,8 +22,10 @@ import {
22
22
  } from "./chunk-F5JBFRAH.js";
23
23
  import {
24
24
  HarnessSourceSelectionSchema,
25
+ createAgentSnapshot,
26
+ createHarnessRelease,
25
27
  validateHarnessSourcePackage
26
- } from "./chunk-UKVK7MRJ.js";
28
+ } from "./chunk-5FBPNWT4.js";
27
29
  import {
28
30
  ImmutableReleaseRefSchema,
29
31
  ReleaseHashSchema,
@@ -214,10 +216,10 @@ function materializeLearningBatchTaskset(input) {
214
216
  const generatedFiles = [];
215
217
  for (const grader of release.graders) {
216
218
  if (grader.kind !== "custom_verifier") continue;
217
- const asset = input.assets?.find((asset2) => asset2.id === grader.verifierRef.id);
218
- if (!asset) throw new Error(`Reward ${grader.id} is missing its immutable verifier source.`);
219
+ const asset2 = input.assets?.find((asset3) => asset3.id === grader.verifierRef.id);
220
+ if (!asset2) throw new Error(`Reward ${grader.id} is missing its immutable verifier source.`);
219
221
  const module = learningVerifierModule(grader.verifierRef.contentHash);
220
- if (!generatedFiles.some((file) => file.path === module)) generatedFiles.push({ path: module, role: "verifier", content: verifyLearningTextAsset(asset, grader.verifierRef) });
222
+ if (!generatedFiles.some((file) => file.path === module)) generatedFiles.push({ path: module, role: "verifier", content: verifyLearningTextAsset(asset2, grader.verifierRef) });
221
223
  }
222
224
  const taskset = TasksetSchema.parse({
223
225
  schemaVersion: "openpond.taskset.v1",
@@ -371,6 +373,269 @@ function verifyFiles(bundle, files) {
371
373
  });
372
374
  }
373
375
 
376
+ // src/training-recipe-binding.ts
377
+ import { z as z3 } from "zod";
378
+ var AdamwOptimizerConfigSchema = z3.object({
379
+ name: z3.literal("adamw").default("adamw"),
380
+ weightDecay: z3.number().nonnegative().max(1).default(0),
381
+ beta1: z3.number().positive().lt(1).default(0.9),
382
+ beta2: z3.number().positive().lt(1).default(0.999),
383
+ epsilon: z3.number().positive().max(0.01).default(1e-8)
384
+ }).strict().default({
385
+ name: "adamw",
386
+ weightDecay: 0,
387
+ beta1: 0.9,
388
+ beta2: 0.999,
389
+ epsilon: 1e-8
390
+ });
391
+ function withAuthoritativeRecipeHashes(taskset, recipe) {
392
+ if (!recipe || typeof recipe !== "object" || Array.isArray(recipe)) {
393
+ return recipe;
394
+ }
395
+ const candidate = recipe;
396
+ if (candidate.method === "dpo") {
397
+ const policyModel = record(candidate.policyModel);
398
+ const referenceModel = record(candidate.referenceModel);
399
+ const dataset2 = record(candidate.dataset);
400
+ const invalidationHash = contentHash({
401
+ tasksetHash: taskset.contentHash,
402
+ preferenceSignals: taskset.learningSignals.preferences.map((signal) => ({
403
+ id: signal.id,
404
+ artifactRef: signal.artifactRef,
405
+ prompt: signal.prompt,
406
+ chosen: signal.chosen,
407
+ rejected: signal.rejected,
408
+ approved: signal.approved
409
+ })),
410
+ policyModel,
411
+ referenceModel,
412
+ dataset: dataset2
413
+ });
414
+ return {
415
+ ...candidate,
416
+ referenceLogprobs: {
417
+ cacheSchemaVersion: "openpond.dpoReferenceLogprobs.v1",
418
+ cacheKey: contentHash(["dpo-reference-logprobs", invalidationHash]),
419
+ invalidationHash
420
+ }
421
+ };
422
+ }
423
+ if (candidate.method === "ppo") {
424
+ const policyOptimization = record(candidate.policyOptimization);
425
+ const policyModel = record(policyOptimization.policyModel);
426
+ const referenceModel = record(policyOptimization.referenceModel);
427
+ const optimizer2 = record(policyOptimization.optimizer);
428
+ const valueModel = record(optimizer2.valueModel);
429
+ const dataset2 = record(policyOptimization.dataset);
430
+ const reward2 = record(policyOptimization.reward);
431
+ const policyHash = contentHash(policyModel);
432
+ const referenceHash = contentHash(referenceModel);
433
+ const valueModelHash = contentHash(valueModel);
434
+ return {
435
+ ...candidate,
436
+ policyOptimization: {
437
+ ...policyOptimization,
438
+ dataset: {
439
+ ...dataset2,
440
+ tasksetId: taskset.id,
441
+ tasksetHash: taskset.contentHash
442
+ },
443
+ reward: {
444
+ ...reward2,
445
+ graderHash: contentHash(taskset.graders)
446
+ }
447
+ },
448
+ resume: {
449
+ ...record(candidate.resume),
450
+ policyHash,
451
+ referenceHash,
452
+ valueModelHash
453
+ }
454
+ };
455
+ }
456
+ if (candidate.method !== "grpo") return recipe;
457
+ const reward = candidate.reward && typeof candidate.reward === "object" && !Array.isArray(candidate.reward) ? candidate.reward : {};
458
+ const metadataToolContractHash = taskset.environment.metadata.toolContractHash;
459
+ const authoritativeToolContractHash = typeof metadataToolContractHash === "string" && metadataToolContractHash.trim() ? metadataToolContractHash : reward.toolContractHash;
460
+ const baseModel = record(candidate.baseModel);
461
+ const dataset = record(candidate.dataset);
462
+ const rollout = record(candidate.rollout);
463
+ const optimizer = record(candidate.optimizer);
464
+ const loss = record(candidate.loss);
465
+ const resourceLimits = record(candidate.resourceLimits);
466
+ const maxExamples = positiveInteger(dataset.maxExamples, 1);
467
+ const groupSize = positiveInteger(rollout.groupSize, 2);
468
+ const maxOutputTokens = positiveInteger(rollout.maxOutputTokens, 1);
469
+ const maxPromptTokens = positiveInteger(dataset.maxPromptTokens, 1);
470
+ const maxSteps = positiveInteger(optimizer.maxSteps, 1);
471
+ const optimizerIterations = positiveInteger(optimizer.iterations, 2);
472
+ return {
473
+ ...candidate,
474
+ resourceLimits: {
475
+ ...resourceLimits,
476
+ maxGpuSeconds: resourceLimits.maxGpuSeconds ?? Math.min(
477
+ 10800,
478
+ Math.ceil(positiveInteger(resourceLimits.wallTimeMs, 18e4) / 1e3)
479
+ )
480
+ },
481
+ reward: {
482
+ ...reward,
483
+ graderHash: contentHash(taskset.graders),
484
+ toolContractHash: authoritativeToolContractHash
485
+ },
486
+ policyOptimization: {
487
+ schemaVersion: "openpond.policyOptimization.v1",
488
+ policyModel: baseModel,
489
+ referenceModel: baseModel,
490
+ dataset: {
491
+ tasksetId: taskset.id,
492
+ tasksetHash: taskset.contentHash,
493
+ split: "train",
494
+ selectionStrategy: dataset.selectionStrategy,
495
+ selectionSeed: rollout.seed,
496
+ maxExamples
497
+ },
498
+ sampler: {
499
+ temperature: rollout.temperature,
500
+ topP: rollout.topP,
501
+ maxOutputTokens,
502
+ maxTurns: rollout.maxTurns,
503
+ concurrency: rollout.concurrency
504
+ },
505
+ environment: {
506
+ id: reward.environmentId,
507
+ version: reward.environmentVersion,
508
+ toolContractHash: authoritativeToolContractHash
509
+ },
510
+ reward: {
511
+ graderId: reward.graderId,
512
+ graderHash: contentHash(taskset.graders),
513
+ learnedPreference: reward.learnedPreference ?? null
514
+ },
515
+ kl: {
516
+ coefficient: loss.klBeta ?? null,
517
+ referenceConstraint: "fixed_reference"
518
+ },
519
+ budgets: {
520
+ maxRollouts: positiveInteger(resourceLimits.maxRollouts, maxExamples * groupSize),
521
+ maxEnvironmentExecutions: positiveInteger(resourceLimits.maxRollouts, maxExamples * groupSize),
522
+ maxInputTokens: maxExamples * groupSize * maxPromptTokens,
523
+ maxOutputTokens: maxExamples * groupSize * maxOutputTokens,
524
+ maxOptimizerSteps: maxSteps * optimizerIterations,
525
+ wallTimeMs: positiveInteger(resourceLimits.wallTimeMs, 18e4),
526
+ maximumCostUsd: null
527
+ },
528
+ checkpointEverySteps: 1,
529
+ seed: rollout.seed,
530
+ evaluationSplit: "frozen_eval",
531
+ optimizer: {
532
+ method: "grpo",
533
+ groupSize,
534
+ normalization: "group_standardized",
535
+ advantageEpsilon: typeof optimizer.advantageEpsilon === "number" ? optimizer.advantageEpsilon : 1e-8,
536
+ loss: loss.method ?? "grpo",
537
+ clipRange: typeof optimizer.clipRange === "number" ? optimizer.clipRange : 0.2,
538
+ iterations: optimizerIterations,
539
+ microbatchSize: positiveInteger(optimizer.microbatchSize, 1),
540
+ gradientAccumulationSteps: positiveInteger(
541
+ optimizer.gradientAccumulationSteps,
542
+ 1
543
+ ),
544
+ adamw: AdamwOptimizerConfigSchema.parse(optimizer.adamw)
545
+ }
546
+ }
547
+ };
548
+ }
549
+ function record(value) {
550
+ return value && typeof value === "object" && !Array.isArray(value) ? value : {};
551
+ }
552
+ function positiveInteger(value, fallback) {
553
+ return typeof value === "number" && Number.isInteger(value) && value > 0 ? value : fallback;
554
+ }
555
+
556
+ // src/training-policy-harness.ts
557
+ function createPolicyHarnessContext(input) {
558
+ const sourceRelease = input.sourceRelease;
559
+ const harnessTools = [];
560
+ const skills = input.skills ?? [];
561
+ const agents = input.agents ?? [];
562
+ if ([...skills, ...agents].some((entry) => entry.visibility !== "policy")) {
563
+ throw new Error("The portable policy Harness requires policy-visible dependencies.");
564
+ }
565
+ const dependencyLock = asset({
566
+ id: "desktop-dependency-lock",
567
+ path: ".openpond/harness/dependency-lock.json",
568
+ hashInput: {
569
+ profileHead: input.profileHead ?? null,
570
+ sourceRelease,
571
+ tools: harnessTools,
572
+ skills: skills.map(({ id, contentHash: assetHash }) => ({ id, contentHash: assetHash })),
573
+ agents: agents.map(({ id, contentHash: assetHash }) => ({ id, contentHash: assetHash }))
574
+ },
575
+ mediaType: "application/json",
576
+ visibility: "policy"
577
+ });
578
+ const agentSnapshot = createAgentSnapshot({
579
+ schemaVersion: "openpond.agentSnapshot.v2",
580
+ id: `agent-snapshot-${contentHash([sourceRelease, harnessTools, dependencyLock.contentHash]).slice(0, 24)}`,
581
+ sourceRelease,
582
+ instructions: [],
583
+ skills,
584
+ agents,
585
+ toolDeclarations: harnessTools,
586
+ capabilityRequirements: [],
587
+ dependencyLock,
588
+ portability: {
589
+ portable: true,
590
+ blockers: [],
591
+ localOnlyAssetRefs: [],
592
+ hostPrivateAssetRefs: []
593
+ },
594
+ metadata: { sourceReleaseId: sourceRelease?.id ?? null }
595
+ });
596
+ const program = asset({
597
+ id: "desktop-harness-program",
598
+ path: ".openpond/harness/program.json",
599
+ hashInput: { program: "openpond.desktop-agent-loop.v1" },
600
+ mediaType: "application/json",
601
+ visibility: "policy"
602
+ });
603
+ const harnessRelease = createHarnessRelease({
604
+ schemaVersion: "openpond.harnessRelease.v2",
605
+ id: `harness-${contentHash([agentSnapshot.contentHash, program.contentHash, harnessTools]).slice(0, 24)}`,
606
+ agentSnapshot: { id: agentSnapshot.id, contentHash: agentSnapshot.contentHash },
607
+ program,
608
+ tools: harnessTools,
609
+ lifecycle: {
610
+ create: true,
611
+ reset: true,
612
+ step: true,
613
+ collect: true,
614
+ destroy: true,
615
+ resetScope: "attempt"
616
+ },
617
+ graderInterface: {
618
+ visibleEvidence: ["output", "runtime_events", "artifacts"],
619
+ privilegedEvidence: ["expected_output", "private_verifier"],
620
+ privateVerifierIsolation: true
621
+ },
622
+ files: [...skills, ...agents],
623
+ metadata: { runtimeProtocol: "openpond.desktop-agent-loop.v1" }
624
+ });
625
+ return { agentSnapshot, harnessRelease };
626
+ }
627
+ function asset(input) {
628
+ const bytes2 = canonicalJson(input.hashInput);
629
+ return {
630
+ id: input.id,
631
+ path: input.path,
632
+ contentHash: contentHash(input.hashInput),
633
+ sizeBytes: new TextEncoder().encode(bytes2).byteLength,
634
+ mediaType: input.mediaType,
635
+ visibility: input.visibility
636
+ };
637
+ }
638
+
374
639
  // src/training-bundle.ts
375
640
  function buildTasksetTrainingBundle(input) {
376
641
  const { taskset, modelProject, harnessRelease, tasksetRelease } = input;
@@ -435,11 +700,11 @@ function buildTasksetTrainingBundle(input) {
435
700
  const verifierAssets = input.verifierAssets ?? [];
436
701
  const references = compileBoundGraders(rewardExecution.binding, rewardExecution.rewards).flatMap((grader) => grader.kind === "custom_verifier" ? [grader.verifierRef] : []);
437
702
  const expected = new Set(references.map((reference) => reference.id));
438
- if (new Set(verifierAssets.map((asset) => asset.id)).size !== verifierAssets.length || verifierAssets.some((asset) => !expected.has(asset.id))) throw new Error("Unexpected or duplicate private verifier asset.");
703
+ if (new Set(verifierAssets.map((asset2) => asset2.id)).size !== verifierAssets.length || verifierAssets.some((asset2) => !expected.has(asset2.id))) throw new Error("Unexpected or duplicate private verifier asset.");
439
704
  for (const reference of references) {
440
- const asset = verifierAssets.find((asset2) => asset2.id === reference.id);
441
- if (!asset || reference.visibility !== "verifier") throw new Error("Training bundle requires its private verifier asset.");
442
- verifyLearningTextAsset2(asset, reference);
705
+ const asset2 = verifierAssets.find((asset3) => asset3.id === reference.id);
706
+ if (!asset2 || reference.visibility !== "verifier") throw new Error("Training bundle requires its private verifier asset.");
707
+ verifyLearningTextAsset2(asset2, reference);
443
708
  }
444
709
  addJsonAsset(assets, "reward-binding.json", { kind: "reward_binding_v1", binding: rewardExecution.binding, rewards: rewardExecution.rewards, assets: verifierAssets });
445
710
  } else if (input.verifierAssets?.length) {
@@ -565,25 +830,25 @@ function buildTasksetTrainingBundle(input) {
565
830
  function addTasksetAssets(input) {
566
831
  const expected = /* @__PURE__ */ new Set();
567
832
  for (const task of input.tasks) {
568
- for (const asset of task.assets ?? []) {
569
- if (expected.has(asset.artifactRef)) {
833
+ for (const asset2 of task.assets ?? []) {
834
+ if (expected.has(asset2.artifactRef)) {
570
835
  throw new Error(
571
- `Training asset path ${asset.artifactRef} is shared by multiple tasks.`
836
+ `Training asset path ${asset2.artifactRef} is shared by multiple tasks.`
572
837
  );
573
838
  }
574
- expected.add(asset.artifactRef);
575
- const value = input.tasksetAssetBytes.get(asset.artifactRef);
839
+ expected.add(asset2.artifactRef);
840
+ const value = input.tasksetAssetBytes.get(asset2.artifactRef);
576
841
  if (!value) {
577
842
  throw new Error(
578
- `Resolved Training Bundle is missing Work asset ${asset.artifactRef}.`
843
+ `Resolved Training Bundle is missing Work asset ${asset2.artifactRef}.`
579
844
  );
580
845
  }
581
- if (value.byteLength !== asset.sizeBytes || sha256(value) !== asset.sha256) {
846
+ if (value.byteLength !== asset2.sizeBytes || sha256(value) !== asset2.sha256) {
582
847
  throw new Error(
583
- `Resolved Training Bundle Work asset ${asset.artifactRef} failed immutable verification.`
848
+ `Resolved Training Bundle Work asset ${asset2.artifactRef} failed immutable verification.`
584
849
  );
585
850
  }
586
- input.assets.set(asset.artifactRef, value);
851
+ input.assets.set(asset2.artifactRef, value);
587
852
  }
588
853
  }
589
854
  for (const assetPath of input.tasksetAssetBytes.keys()) {
@@ -601,6 +866,7 @@ function bytes(value) {
601
866
  return Buffer.from(canonicalJson(value), "utf8");
602
867
  }
603
868
  export {
869
+ AdamwOptimizerConfigSchema,
604
870
  ComputeTargetBindingSchema,
605
871
  HarnessRunManifestContentSchema,
606
872
  HarnessRunManifestSchema,
@@ -611,9 +877,11 @@ export {
611
877
  ScopedSecretDeclarationSchema,
612
878
  TrainingEngineBindingSchema,
613
879
  buildTasksetTrainingBundle,
880
+ createPolicyHarnessContext,
614
881
  materializeLearningBatchTaskset,
615
882
  prepareManagedTrainingSubmission,
616
883
  prepareReviewedLearningBatch,
617
- scanAndRedactEvidence
884
+ scanAndRedactEvidence,
885
+ withAuthoritativeRecipeHashes
618
886
  };
619
887
  //# sourceMappingURL=training-bundle.js.map