openpond-sdk 0.2.4 → 0.2.5
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/TASKSETS.md +30 -0
- package/dist/model-batch-review.js +48 -0
- package/dist/model-batch-review.js.map +7 -0
- package/dist/taskset-packages.js +291 -0
- package/dist/taskset-packages.js.map +4 -4
- package/dist/types/packages/sdk/src/model-batch-review-contracts.d.ts +568 -0
- package/dist/types/packages/sdk/src/model-batch-review-contracts.d.ts.map +1 -0
- package/dist/types/packages/sdk/src/model-batch-review-preparation.d.ts +351 -0
- package/dist/types/packages/sdk/src/model-batch-review-preparation.d.ts.map +1 -0
- package/dist/types/packages/sdk/src/model-batch-review.d.ts +543 -0
- package/dist/types/packages/sdk/src/model-batch-review.d.ts.map +1 -0
- package/dist/types/packages/sdk/src/model-projects.d.ts +2 -2
- package/dist/types/packages/sdk/src/taskset-package-client.d.ts +38 -0
- package/dist/types/packages/sdk/src/taskset-package-client.d.ts.map +1 -1
- package/dist/types/packages/sdk/src/taskset-package-contracts.d.ts +38 -0
- package/dist/types/packages/sdk/src/taskset-package-contracts.d.ts.map +1 -1
- package/dist/types/packages/sdk/src/taskset-package-learning.d.ts +38 -0
- package/dist/types/packages/sdk/src/taskset-package-learning.d.ts.map +1 -1
- package/dist/types/packages/sdk/src/taskset-packages.d.ts +1 -0
- package/dist/types/packages/sdk/src/taskset-packages.d.ts.map +1 -1
- package/package.json +6 -2
package/TASKSETS.md
CHANGED
|
@@ -86,3 +86,33 @@ release/package hashes; Profile provenance makes these different identities.
|
|
|
86
86
|
Persist the exact publication intent and immutable package before sending it.
|
|
87
87
|
After an uncertain response, replay that intent before publishing later local
|
|
88
88
|
edits, then merge the recovered hosted receipt without reverting those edits.
|
|
89
|
+
|
|
90
|
+
## Revising an approved batch
|
|
91
|
+
|
|
92
|
+
Browser editors import schemas and types from `openpond-sdk/model-batch-review`.
|
|
93
|
+
Servers use `inspectModelBatchPackage` from `openpond-sdk/taskset-packages` to
|
|
94
|
+
validate the complete package and return its definition, binding, evidence, and
|
|
95
|
+
decisions without portable file bytes. The inspection includes the sealed package
|
|
96
|
+
hash. Full package validation and compilation remain on the server.
|
|
97
|
+
|
|
98
|
+
`ModelBatchReviewRequestSchema`, `findModelBatchReview`, and
|
|
99
|
+
`beginModelBatchReview` define explicit editing of a Model's selected reviewed
|
|
100
|
+
batch. The host runs these helpers inside its authorized workspace transaction,
|
|
101
|
+
checks the Model revision and selected Taskset reference, and verifies that the
|
|
102
|
+
supplied package belongs to that immutable selection. An operation ID identifies
|
|
103
|
+
one exact request; retries return the original receipt, while changed requests
|
|
104
|
+
with the same operation ID conflict.
|
|
105
|
+
|
|
106
|
+
The request can change task instructions, schemas, inputs, evaluator-only context,
|
|
107
|
+
expected answers, and the selected published Reward binding. The helper creates
|
|
108
|
+
a new definition, Reward graph, enabled direct source, and evidence with parent
|
|
109
|
+
references. Sources record the original Model, package, batch, and Reward binding
|
|
110
|
+
in `reviewOrigin`. Original source credentials and admission decisions are never
|
|
111
|
+
imported. Attempts receive new IDs so matching example/attempt IDs from different
|
|
112
|
+
parent sources remain distinct.
|
|
113
|
+
|
|
114
|
+
Previously approved targets become pending feedback proposals. Policy-visible
|
|
115
|
+
task changes clear the observed response because that response was produced for
|
|
116
|
+
the original task. New evidence needs fresh grading and review before sealing;
|
|
117
|
+
the helper does not attach a batch or start training. A host must separately
|
|
118
|
+
compare the Model revision when attaching the newly prepared batch.
|
|
@@ -0,0 +1,48 @@
|
|
|
1
|
+
// src/model-batch-review-contracts.ts
|
|
2
|
+
import { z } from "zod";
|
|
3
|
+
import { LearningJsonObjectSchema, LearningRevisionRefSchema, LearningSourceSchema, TaskEvidenceSchema, TaskFeedbackSchema, TaskDefinitionSchema, TaskAdmissionDecisionSchema } from "@openpond/evals/learning";
|
|
4
|
+
import { RewardBindingSchema } from "@openpond/evals/rewards";
|
|
5
|
+
var ReleaseIdSchema = LearningRevisionRefSchema.shape.id;
|
|
6
|
+
var ModelBatchReviewInspectionSchema = z.object({
|
|
7
|
+
schemaVersion: z.literal("openpond.modelBatchReviewInspection.v1"),
|
|
8
|
+
packageHash: z.string().regex(/^[a-f0-9]{64}$/),
|
|
9
|
+
definition: TaskDefinitionSchema,
|
|
10
|
+
binding: RewardBindingSchema,
|
|
11
|
+
evidence: z.array(TaskEvidenceSchema).min(1).max(1e4),
|
|
12
|
+
decisions: z.array(TaskAdmissionDecisionSchema).min(1).max(1e4)
|
|
13
|
+
}).strict();
|
|
14
|
+
var ModelBatchReviewRequestSchema = z.object({
|
|
15
|
+
schemaVersion: z.literal("openpond.modelBatchReviewRequest.v1"),
|
|
16
|
+
operationId: ReleaseIdSchema,
|
|
17
|
+
modelId: ReleaseIdSchema,
|
|
18
|
+
expectedModelRevision: z.number().int().positive(),
|
|
19
|
+
tasksetRef: LearningRevisionRefSchema,
|
|
20
|
+
rewardBindingRef: LearningRevisionRefSchema.nullable(),
|
|
21
|
+
definition: z.object({
|
|
22
|
+
name: z.string().trim().min(1).max(500).optional(),
|
|
23
|
+
instructions: z.string().trim().min(1).max(2e4).optional(),
|
|
24
|
+
inputSchema: LearningJsonObjectSchema.optional(),
|
|
25
|
+
outputSchema: LearningJsonObjectSchema.optional()
|
|
26
|
+
}).strict(),
|
|
27
|
+
examples: z.array(z.object({
|
|
28
|
+
evidence: LearningRevisionRefSchema,
|
|
29
|
+
input: LearningJsonObjectSchema.optional(),
|
|
30
|
+
expected: LearningJsonObjectSchema.nullable().optional(),
|
|
31
|
+
evaluatorContext: LearningJsonObjectSchema.nullable().optional(),
|
|
32
|
+
proposedTarget: LearningJsonObjectSchema.nullable().optional()
|
|
33
|
+
}).strict()).max(1e4)
|
|
34
|
+
}).strict();
|
|
35
|
+
var ModelBatchReviewReceiptSchema = z.object({
|
|
36
|
+
schemaVersion: z.literal("openpond.modelBatchReviewReceipt.v1"),
|
|
37
|
+
operationId: ReleaseIdSchema,
|
|
38
|
+
modelId: ReleaseIdSchema,
|
|
39
|
+
source: LearningSourceSchema,
|
|
40
|
+
evidence: z.array(TaskEvidenceSchema).min(1).max(1e4),
|
|
41
|
+
proposals: z.array(TaskFeedbackSchema).max(1e4)
|
|
42
|
+
}).strict();
|
|
43
|
+
export {
|
|
44
|
+
ModelBatchReviewInspectionSchema,
|
|
45
|
+
ModelBatchReviewReceiptSchema,
|
|
46
|
+
ModelBatchReviewRequestSchema
|
|
47
|
+
};
|
|
48
|
+
//# sourceMappingURL=model-batch-review.js.map
|
|
@@ -0,0 +1,7 @@
|
|
|
1
|
+
{
|
|
2
|
+
"version": 3,
|
|
3
|
+
"sources": ["../src/model-batch-review-contracts.ts"],
|
|
4
|
+
"sourcesContent": ["import { z } from \"zod\";\nimport { LearningJsonObjectSchema, LearningRevisionRefSchema, LearningSourceSchema, TaskEvidenceSchema, TaskFeedbackSchema, TaskDefinitionSchema, TaskAdmissionDecisionSchema } from \"@openpond/evals/learning\";\nimport { RewardBindingSchema } from \"@openpond/evals/rewards\";\n// Reuse the canonical release ID already exposed by Evals without bundling a\n// second copy of Harness's schema graph into the browser contract entrypoint.\nconst ReleaseIdSchema = LearningRevisionRefSchema.shape.id;\n\n/** Editor data from a server-validated package; excludes portable file bytes. */\nexport const ModelBatchReviewInspectionSchema = z.object({\n schemaVersion: z.literal(\"openpond.modelBatchReviewInspection.v1\"),\n packageHash: z.string().regex(/^[a-f0-9]{64}$/),\n definition: TaskDefinitionSchema,\n binding: RewardBindingSchema,\n evidence: z.array(TaskEvidenceSchema).min(1).max(10_000),\n decisions: z.array(TaskAdmissionDecisionSchema).min(1).max(10_000),\n}).strict();\nexport type ModelBatchReviewInspection = z.infer<typeof ModelBatchReviewInspectionSchema>;\n\nexport const ModelBatchReviewRequestSchema = z.object({\n schemaVersion: z.literal(\"openpond.modelBatchReviewRequest.v1\"),\n operationId: ReleaseIdSchema,\n modelId: ReleaseIdSchema,\n expectedModelRevision: z.number().int().positive(),\n tasksetRef: LearningRevisionRefSchema,\n rewardBindingRef: LearningRevisionRefSchema.nullable(),\n definition: z.object({\n name: z.string().trim().min(1).max(500).optional(),\n instructions: z.string().trim().min(1).max(20_000).optional(),\n inputSchema: LearningJsonObjectSchema.optional(),\n outputSchema: LearningJsonObjectSchema.optional(),\n }).strict(),\n examples: z.array(z.object({\n evidence: LearningRevisionRefSchema,\n input: LearningJsonObjectSchema.optional(),\n expected: LearningJsonObjectSchema.nullable().optional(),\n evaluatorContext: LearningJsonObjectSchema.nullable().optional(),\n proposedTarget: LearningJsonObjectSchema.nullable().optional(),\n }).strict()).max(10_000),\n}).strict();\nexport type ModelBatchReviewRequest = z.infer<typeof ModelBatchReviewRequestSchema>;\n\n/** These are editable evidence and unapproved proposals, never admissions. */\nexport const ModelBatchReviewReceiptSchema = z.object({\n schemaVersion: z.literal(\"openpond.modelBatchReviewReceipt.v1\"),\n operationId: ReleaseIdSchema,\n modelId: ReleaseIdSchema,\n source: LearningSourceSchema,\n evidence: z.array(TaskEvidenceSchema).min(1).max(10_000),\n proposals: z.array(TaskFeedbackSchema).max(10_000),\n}).strict();\nexport type ModelBatchReviewReceipt = z.infer<typeof ModelBatchReviewReceiptSchema>;\n"],
|
|
5
|
+
"mappings": ";AAAA,SAAS,SAAS;AAClB,SAAS,0BAA0B,2BAA2B,sBAAsB,oBAAoB,oBAAoB,sBAAsB,mCAAmC;AACrL,SAAS,2BAA2B;AAGpC,IAAM,kBAAkB,0BAA0B,MAAM;AAGjD,IAAM,mCAAmC,EAAE,OAAO;AAAA,EACvD,eAAe,EAAE,QAAQ,wCAAwC;AAAA,EACjE,aAAa,EAAE,OAAO,EAAE,MAAM,gBAAgB;AAAA,EAC9C,YAAY;AAAA,EACZ,SAAS;AAAA,EACT,UAAU,EAAE,MAAM,kBAAkB,EAAE,IAAI,CAAC,EAAE,IAAI,GAAM;AAAA,EACvD,WAAW,EAAE,MAAM,2BAA2B,EAAE,IAAI,CAAC,EAAE,IAAI,GAAM;AACnE,CAAC,EAAE,OAAO;AAGH,IAAM,gCAAgC,EAAE,OAAO;AAAA,EACpD,eAAe,EAAE,QAAQ,qCAAqC;AAAA,EAC9D,aAAa;AAAA,EACb,SAAS;AAAA,EACT,uBAAuB,EAAE,OAAO,EAAE,IAAI,EAAE,SAAS;AAAA,EACjD,YAAY;AAAA,EACZ,kBAAkB,0BAA0B,SAAS;AAAA,EACrD,YAAY,EAAE,OAAO;AAAA,IACnB,MAAM,EAAE,OAAO,EAAE,KAAK,EAAE,IAAI,CAAC,EAAE,IAAI,GAAG,EAAE,SAAS;AAAA,IACjD,cAAc,EAAE,OAAO,EAAE,KAAK,EAAE,IAAI,CAAC,EAAE,IAAI,GAAM,EAAE,SAAS;AAAA,IAC5D,aAAa,yBAAyB,SAAS;AAAA,IAC/C,cAAc,yBAAyB,SAAS;AAAA,EAClD,CAAC,EAAE,OAAO;AAAA,EACV,UAAU,EAAE,MAAM,EAAE,OAAO;AAAA,IACzB,UAAU;AAAA,IACV,OAAO,yBAAyB,SAAS;AAAA,IACzC,UAAU,yBAAyB,SAAS,EAAE,SAAS;AAAA,IACvD,kBAAkB,yBAAyB,SAAS,EAAE,SAAS;AAAA,IAC/D,gBAAgB,yBAAyB,SAAS,EAAE,SAAS;AAAA,EAC/D,CAAC,EAAE,OAAO,CAAC,EAAE,IAAI,GAAM;AACzB,CAAC,EAAE,OAAO;AAIH,IAAM,gCAAgC,EAAE,OAAO;AAAA,EACpD,eAAe,EAAE,QAAQ,qCAAqC;AAAA,EAC9D,aAAa;AAAA,EACb,SAAS;AAAA,EACT,QAAQ;AAAA,EACR,UAAU,EAAE,MAAM,kBAAkB,EAAE,IAAI,CAAC,EAAE,IAAI,GAAM;AAAA,EACvD,WAAW,EAAE,MAAM,kBAAkB,EAAE,IAAI,GAAM;AACnD,CAAC,EAAE,OAAO;",
|
|
6
|
+
"names": []
|
|
7
|
+
}
|
package/dist/taskset-packages.js
CHANGED
|
@@ -2864,8 +2864,296 @@ var OpenPondTasksetPackageClient = class {
|
|
|
2864
2864
|
return payload;
|
|
2865
2865
|
}
|
|
2866
2866
|
};
|
|
2867
|
+
|
|
2868
|
+
// src/model-batch-review.ts
|
|
2869
|
+
import {
|
|
2870
|
+
LearningDomainError as LearningDomainError2,
|
|
2871
|
+
learningRef as learningRef6,
|
|
2872
|
+
sameLearningRef as sameLearningRef4,
|
|
2873
|
+
requireLearningRelease,
|
|
2874
|
+
requireLearningResource,
|
|
2875
|
+
taskBatchPackageMetadata as taskBatchPackageMetadata4
|
|
2876
|
+
} from "@openpond/evals/learning";
|
|
2877
|
+
|
|
2878
|
+
// src/model-batch-review-contracts.ts
|
|
2879
|
+
import { z as z20 } from "zod";
|
|
2880
|
+
import { LearningJsonObjectSchema, LearningRevisionRefSchema, LearningSourceSchema as LearningSourceSchema2, TaskEvidenceSchema as TaskEvidenceSchema2, TaskFeedbackSchema, TaskDefinitionSchema as TaskDefinitionSchema2, TaskAdmissionDecisionSchema as TaskAdmissionDecisionSchema2 } from "@openpond/evals/learning";
|
|
2881
|
+
import { RewardBindingSchema as RewardBindingSchema2 } from "@openpond/evals/rewards";
|
|
2882
|
+
var ReleaseIdSchema2 = LearningRevisionRefSchema.shape.id;
|
|
2883
|
+
var ModelBatchReviewInspectionSchema = z20.object({
|
|
2884
|
+
schemaVersion: z20.literal("openpond.modelBatchReviewInspection.v1"),
|
|
2885
|
+
packageHash: z20.string().regex(/^[a-f0-9]{64}$/),
|
|
2886
|
+
definition: TaskDefinitionSchema2,
|
|
2887
|
+
binding: RewardBindingSchema2,
|
|
2888
|
+
evidence: z20.array(TaskEvidenceSchema2).min(1).max(1e4),
|
|
2889
|
+
decisions: z20.array(TaskAdmissionDecisionSchema2).min(1).max(1e4)
|
|
2890
|
+
}).strict();
|
|
2891
|
+
var ModelBatchReviewRequestSchema = z20.object({
|
|
2892
|
+
schemaVersion: z20.literal("openpond.modelBatchReviewRequest.v1"),
|
|
2893
|
+
operationId: ReleaseIdSchema2,
|
|
2894
|
+
modelId: ReleaseIdSchema2,
|
|
2895
|
+
expectedModelRevision: z20.number().int().positive(),
|
|
2896
|
+
tasksetRef: LearningRevisionRefSchema,
|
|
2897
|
+
rewardBindingRef: LearningRevisionRefSchema.nullable(),
|
|
2898
|
+
definition: z20.object({
|
|
2899
|
+
name: z20.string().trim().min(1).max(500).optional(),
|
|
2900
|
+
instructions: z20.string().trim().min(1).max(2e4).optional(),
|
|
2901
|
+
inputSchema: LearningJsonObjectSchema.optional(),
|
|
2902
|
+
outputSchema: LearningJsonObjectSchema.optional()
|
|
2903
|
+
}).strict(),
|
|
2904
|
+
examples: z20.array(z20.object({
|
|
2905
|
+
evidence: LearningRevisionRefSchema,
|
|
2906
|
+
input: LearningJsonObjectSchema.optional(),
|
|
2907
|
+
expected: LearningJsonObjectSchema.nullable().optional(),
|
|
2908
|
+
evaluatorContext: LearningJsonObjectSchema.nullable().optional(),
|
|
2909
|
+
proposedTarget: LearningJsonObjectSchema.nullable().optional()
|
|
2910
|
+
}).strict()).max(1e4)
|
|
2911
|
+
}).strict();
|
|
2912
|
+
var ModelBatchReviewReceiptSchema = z20.object({
|
|
2913
|
+
schemaVersion: z20.literal("openpond.modelBatchReviewReceipt.v1"),
|
|
2914
|
+
operationId: ReleaseIdSchema2,
|
|
2915
|
+
modelId: ReleaseIdSchema2,
|
|
2916
|
+
source: LearningSourceSchema2,
|
|
2917
|
+
evidence: z20.array(TaskEvidenceSchema2).min(1).max(1e4),
|
|
2918
|
+
proposals: z20.array(TaskFeedbackSchema).max(1e4)
|
|
2919
|
+
}).strict();
|
|
2920
|
+
|
|
2921
|
+
// src/model-batch-review-preparation.ts
|
|
2922
|
+
import {
|
|
2923
|
+
LearningDomainError,
|
|
2924
|
+
LearningSourceSchema as LearningSourceSchema3,
|
|
2925
|
+
TaskDefinitionSchema as TaskDefinitionSchema3,
|
|
2926
|
+
TaskEvidenceSchema as TaskEvidenceSchema3,
|
|
2927
|
+
TaskFeedbackSchema as TaskFeedbackSchema2,
|
|
2928
|
+
assertLearningContentHash as assertLearningContentHash3,
|
|
2929
|
+
learningEvidenceId,
|
|
2930
|
+
learningRef as learningRef5,
|
|
2931
|
+
sameLearningRef as sameLearningRef3,
|
|
2932
|
+
sealLearningContent as sealLearningContent4,
|
|
2933
|
+
taskBatchPackageMetadata as taskBatchPackageMetadata3,
|
|
2934
|
+
verifyLearningTextAsset as verifyLearningTextAsset5
|
|
2935
|
+
} from "@openpond/evals/learning";
|
|
2936
|
+
import { compileBoundGraders as compileBoundGraders2, createRewardBinding, createRewardRelease, resolveBoundRewards } from "@openpond/evals/rewards";
|
|
2937
|
+
function prepareModelBatchReview(input) {
|
|
2938
|
+
const { request, package: value } = input;
|
|
2939
|
+
const learning = value.learningResources;
|
|
2940
|
+
if (!learning) throw new LearningDomainError("model_review_batch_required", 422);
|
|
2941
|
+
const metadata = taskBatchPackageMetadata3(value.taskset);
|
|
2942
|
+
const id = `model-review-${contentHash([input.scope, request.modelId, request.operationId])}`;
|
|
2943
|
+
const edits = new Map(request.examples.map((edit) => [contentHash(edit.evidence), edit]));
|
|
2944
|
+
const parentRefs = new Set(learning.evidence.map((evidence2) => contentHash(learningRef5(evidence2))));
|
|
2945
|
+
const decisions = new Map(learning.decisions.map((decision) => [contentHash(decision.evidence), decision]));
|
|
2946
|
+
if (edits.size !== request.examples.length || request.examples.some((edit) => !parentRefs.has(contentHash(edit.evidence)))) {
|
|
2947
|
+
throw new LearningDomainError("model_review_evidence_mismatch", 409);
|
|
2948
|
+
}
|
|
2949
|
+
for (const resource of [input.binding, ...input.rewards, ...input.assets]) assertLearningContentHash3(resource);
|
|
2950
|
+
resolveBoundRewards(input.binding, input.rewards);
|
|
2951
|
+
const rewardByHash = /* @__PURE__ */ new Map();
|
|
2952
|
+
for (const source2 of input.binding.sources) {
|
|
2953
|
+
const original = input.rewards.find((reward) => sameLearningRef3(learningRef5(reward), source2.reward));
|
|
2954
|
+
const { contentHash: _hash, ...content } = original;
|
|
2955
|
+
rewardByHash.set(original.contentHash, createRewardRelease({ ...content, id: `${id}-reward-${original.contentHash}`, revision: 1 }));
|
|
2956
|
+
}
|
|
2957
|
+
const rewards = [...rewardByHash.values()];
|
|
2958
|
+
const { contentHash: _bindingHash, recipeRef: _recipe, ...bindingContent } = input.binding;
|
|
2959
|
+
const binding = createRewardBinding({
|
|
2960
|
+
...bindingContent,
|
|
2961
|
+
id: `${id}-binding`,
|
|
2962
|
+
revision: 1,
|
|
2963
|
+
sources: input.binding.sources.map((source2) => ({ ...source2, reward: learningRef5(rewardByHash.get(source2.reward.contentHash)) }))
|
|
2964
|
+
}, rewards);
|
|
2965
|
+
const assetRefs = rewards.flatMap((reward) => [
|
|
2966
|
+
...reward.assets,
|
|
2967
|
+
..."verifierRef" in reward.implementation ? [reward.implementation.verifierRef] : [],
|
|
2968
|
+
..."rubricRef" in reward.implementation ? [reward.implementation.rubricRef] : [],
|
|
2969
|
+
..."inputContract" in reward.implementation ? [reward.implementation.inputContract] : []
|
|
2970
|
+
]);
|
|
2971
|
+
const assets = [...new Map(assetRefs.map((ref) => {
|
|
2972
|
+
const asset = input.assets.find((asset2) => asset2.id === ref.id);
|
|
2973
|
+
if (!asset) throw new LearningDomainError("model_review_reward_asset_missing", 422);
|
|
2974
|
+
verifyLearningTextAsset5(asset, ref);
|
|
2975
|
+
if (ref.visibility === "policy") throw new LearningDomainError("reward_asset_visibility_invalid", 422);
|
|
2976
|
+
return [asset.id, asset];
|
|
2977
|
+
})).values()];
|
|
2978
|
+
const { environmentRelease: _environment, verifierSetRelease: _verifiers, ...execution } = metadata.definition.execution;
|
|
2979
|
+
const { contentHash: _definitionHash, ...definitionContent } = metadata.definition;
|
|
2980
|
+
const definition = TaskDefinitionSchema3.parse(sealLearningContent4({
|
|
2981
|
+
...definitionContent,
|
|
2982
|
+
...request.definition,
|
|
2983
|
+
id: `${id}-definition`,
|
|
2984
|
+
revision: 1,
|
|
2985
|
+
rewardBinding: learningRef5(binding),
|
|
2986
|
+
execution: {
|
|
2987
|
+
...execution,
|
|
2988
|
+
policy: { ...execution.policy, hiddenGraderRefs: compileBoundGraders2(binding, rewards).filter((grader) => grader.privileged).map((grader) => grader.id) }
|
|
2989
|
+
}
|
|
2990
|
+
}));
|
|
2991
|
+
const source = LearningSourceSchema3.parse(sealLearningContent4({
|
|
2992
|
+
schemaVersion: "openpond.learningSource.v1",
|
|
2993
|
+
id,
|
|
2994
|
+
revision: 1,
|
|
2995
|
+
name: `${definition.name} review`,
|
|
2996
|
+
kind: "direct",
|
|
2997
|
+
taskDefinition: learningRef5(definition),
|
|
2998
|
+
enabled: true,
|
|
2999
|
+
allowedSplits: [...new Set(learning.evidence.map((item) => item.submission.split))],
|
|
3000
|
+
mapping: null,
|
|
3001
|
+
adapterVersion: null,
|
|
3002
|
+
reviewOrigin: {
|
|
3003
|
+
modelId: request.modelId,
|
|
3004
|
+
packageHash: value.contentHash,
|
|
3005
|
+
taskset: learningRef5(value.taskset),
|
|
3006
|
+
batch: learningRef5(learning.batch),
|
|
3007
|
+
rewardBinding: learningRef5(input.binding)
|
|
3008
|
+
}
|
|
3009
|
+
}));
|
|
3010
|
+
const evidence = learning.evidence.map((parent) => {
|
|
3011
|
+
const edit = edits.get(contentHash(learningRef5(parent)));
|
|
3012
|
+
const attemptId = `review-${parent.contentHash}`;
|
|
3013
|
+
const policyChanged = contentHash({
|
|
3014
|
+
instructions: definition.instructions,
|
|
3015
|
+
inputSchema: definition.inputSchema,
|
|
3016
|
+
outputSchema: definition.outputSchema,
|
|
3017
|
+
input: edit?.input ?? parent.submission.input
|
|
3018
|
+
}) !== contentHash({
|
|
3019
|
+
instructions: metadata.definition.instructions,
|
|
3020
|
+
inputSchema: metadata.definition.inputSchema,
|
|
3021
|
+
outputSchema: metadata.definition.outputSchema,
|
|
3022
|
+
input: parent.submission.input
|
|
3023
|
+
});
|
|
3024
|
+
return TaskEvidenceSchema3.parse(sealLearningContent4({
|
|
3025
|
+
schemaVersion: "openpond.taskEvidence.v1",
|
|
3026
|
+
id: learningEvidenceId(source.id, parent.submission.exampleId, attemptId),
|
|
3027
|
+
revision: 1,
|
|
3028
|
+
source: learningRef5(source),
|
|
3029
|
+
submission: {
|
|
3030
|
+
...parent.submission,
|
|
3031
|
+
sourceId: source.id,
|
|
3032
|
+
taskDefinition: learningRef5(definition),
|
|
3033
|
+
attemptId,
|
|
3034
|
+
// The old response was observed under the old policy-visible task. Its
|
|
3035
|
+
// parent snapshot remains available, but it is not a run of new inputs.
|
|
3036
|
+
observedOutput: policyChanged ? null : parent.submission.observedOutput,
|
|
3037
|
+
idempotencyKey: `${id}-${parent.contentHash}`,
|
|
3038
|
+
...edit?.input === void 0 ? {} : { input: edit.input },
|
|
3039
|
+
...edit?.expected === void 0 ? {} : { expected: edit.expected },
|
|
3040
|
+
...edit?.evaluatorContext === void 0 ? {} : { evaluatorContext: edit.evaluatorContext }
|
|
3041
|
+
},
|
|
3042
|
+
supersedes: learningRef5(parent),
|
|
3043
|
+
correctionFeedbackId: null,
|
|
3044
|
+
receivedAt: input.now
|
|
3045
|
+
}));
|
|
3046
|
+
});
|
|
3047
|
+
const proposals = evidence.flatMap((item) => {
|
|
3048
|
+
const edit = edits.get(contentHash(item.supersedes));
|
|
3049
|
+
const decision = decisions.get(contentHash(item.supersedes));
|
|
3050
|
+
const target = edit?.proposedTarget === void 0 ? decision.approvedTarget : edit.proposedTarget;
|
|
3051
|
+
return target === null ? [] : [TaskFeedbackSchema2.parse({
|
|
3052
|
+
schemaVersion: "openpond.taskFeedbackRecord.v1",
|
|
3053
|
+
id: `feedback-${contentHash([id, item.id])}`,
|
|
3054
|
+
revision: 1,
|
|
3055
|
+
submission: {
|
|
3056
|
+
schemaVersion: "openpond.taskFeedback.v1",
|
|
3057
|
+
sourceId: source.id,
|
|
3058
|
+
idempotencyKey: `${id}-${item.id}`,
|
|
3059
|
+
exampleId: item.submission.exampleId,
|
|
3060
|
+
attemptId: item.submission.attemptId,
|
|
3061
|
+
expectedEvidenceHash: item.contentHash,
|
|
3062
|
+
occurredAt: input.now,
|
|
3063
|
+
kind: "target_correction",
|
|
3064
|
+
value: target,
|
|
3065
|
+
note: "Proposed target for the revised task. Grade and review before admission."
|
|
3066
|
+
},
|
|
3067
|
+
status: "pending_review",
|
|
3068
|
+
evidence: learningRef5(item),
|
|
3069
|
+
createdAt: input.now,
|
|
3070
|
+
review: null
|
|
3071
|
+
})];
|
|
3072
|
+
});
|
|
3073
|
+
return { source, definition, binding, rewards, assets, evidence, proposals };
|
|
3074
|
+
}
|
|
3075
|
+
|
|
3076
|
+
// src/model-batch-review.ts
|
|
3077
|
+
import { LearningDomainError as LearningDomainError3 } from "@openpond/evals/learning";
|
|
3078
|
+
var operation = (request) => `model-batch-review-${contentHash([request.modelId, request.operationId])}`;
|
|
3079
|
+
function inspectModelBatchPackage(raw) {
|
|
3080
|
+
const value = validateTasksetPackage(raw);
|
|
3081
|
+
if (!value.learningResources) throw new LearningDomainError2("model_review_batch_required", 422);
|
|
3082
|
+
const metadata = taskBatchPackageMetadata4(value.taskset);
|
|
3083
|
+
return ModelBatchReviewInspectionSchema.parse({
|
|
3084
|
+
schemaVersion: "openpond.modelBatchReviewInspection.v1",
|
|
3085
|
+
packageHash: value.contentHash,
|
|
3086
|
+
definition: metadata.definition,
|
|
3087
|
+
binding: metadata.binding,
|
|
3088
|
+
evidence: value.learningResources.evidence,
|
|
3089
|
+
decisions: value.learningResources.decisions
|
|
3090
|
+
});
|
|
3091
|
+
}
|
|
3092
|
+
async function findModelBatchReview(transaction, raw) {
|
|
3093
|
+
const request = ModelBatchReviewRequestSchema.parse(raw);
|
|
3094
|
+
const prior = await transaction.operation(operation(request));
|
|
3095
|
+
if (!prior) return null;
|
|
3096
|
+
if (prior.requestHash !== contentHash(request)) throw new LearningDomainError2("learning_idempotency_conflict", 409);
|
|
3097
|
+
return readReceipt(transaction, request, prior.resources);
|
|
3098
|
+
}
|
|
3099
|
+
async function beginModelBatchReview(transaction, input) {
|
|
3100
|
+
const request = ModelBatchReviewRequestSchema.parse(input.request);
|
|
3101
|
+
const prior = await findModelBatchReview(transaction, request);
|
|
3102
|
+
if (prior) return prior;
|
|
3103
|
+
const value = validateTasksetPackage(input.package);
|
|
3104
|
+
if (!value.learningResources) throw new LearningDomainError2("model_review_batch_required", 422);
|
|
3105
|
+
const metadata = taskBatchPackageMetadata4(value.taskset);
|
|
3106
|
+
const selected = request.rewardBindingRef;
|
|
3107
|
+
const retained = !selected || sameLearningRef4(selected, learningRef6(metadata.binding));
|
|
3108
|
+
const binding = retained ? metadata.binding : await requireLearningRelease(transaction, "binding", selected);
|
|
3109
|
+
const rewards = retained ? metadata.rewards : await Promise.all(binding.sources.map((source) => requireLearningRelease(transaction, "reward", source.reward)));
|
|
3110
|
+
const assetIds = new Set(rewards.flatMap((reward) => [
|
|
3111
|
+
...reward.assets,
|
|
3112
|
+
..."verifierRef" in reward.implementation ? [reward.implementation.verifierRef] : [],
|
|
3113
|
+
..."rubricRef" in reward.implementation ? [reward.implementation.rubricRef] : [],
|
|
3114
|
+
..."inputContract" in reward.implementation ? [reward.implementation.inputContract] : []
|
|
3115
|
+
]).map((asset) => asset.id));
|
|
3116
|
+
const assets = retained ? value.learningResources.assets : await Promise.all([...assetIds].map((id) => requireLearningResource(transaction, "asset", id, 1)));
|
|
3117
|
+
const prepared = prepareModelBatchReview({ ...input, request, package: value, binding, rewards, assets });
|
|
3118
|
+
const pointers = [];
|
|
3119
|
+
for (const asset of prepared.assets) {
|
|
3120
|
+
const existing = await transaction.get("asset", asset.id, 1);
|
|
3121
|
+
if (existing && existing.contentHash !== asset.contentHash) throw new LearningDomainError2("model_review_asset_conflict", 409);
|
|
3122
|
+
if (!existing) await transaction.put("asset", asset, 0);
|
|
3123
|
+
}
|
|
3124
|
+
for (const reward of prepared.rewards) await transaction.put("reward", reward, 0);
|
|
3125
|
+
await transaction.put("binding", prepared.binding, 0);
|
|
3126
|
+
await transaction.put("definition", prepared.definition, 0);
|
|
3127
|
+
await transaction.put("source", prepared.source, 0, { parentId: request.modelId });
|
|
3128
|
+
pointers.push({ kind: "source", id: prepared.source.id, revision: 1 });
|
|
3129
|
+
for (const evidence of prepared.evidence) {
|
|
3130
|
+
await transaction.put("evidence", evidence, 0, { parentId: prepared.source.id });
|
|
3131
|
+
pointers.push({ kind: "evidence", id: evidence.id, revision: 1 });
|
|
3132
|
+
}
|
|
3133
|
+
for (const proposal of prepared.proposals) {
|
|
3134
|
+
await transaction.put("feedback", proposal, 0, { parentId: proposal.evidence.id, status: proposal.status });
|
|
3135
|
+
pointers.push({ kind: "feedback", id: proposal.id, revision: 1 });
|
|
3136
|
+
}
|
|
3137
|
+
await transaction.saveOperation(operation(request), { requestHash: contentHash(request), resources: pointers });
|
|
3138
|
+
return readReceipt(transaction, request, pointers);
|
|
3139
|
+
}
|
|
3140
|
+
async function readReceipt(transaction, request, pointers) {
|
|
3141
|
+
const resources = await Promise.all(pointers.map((pointer) => requireLearningResource(transaction, pointer.kind, pointer.id, pointer.revision)));
|
|
3142
|
+
return ModelBatchReviewReceiptSchema.parse({
|
|
3143
|
+
schemaVersion: "openpond.modelBatchReviewReceipt.v1",
|
|
3144
|
+
operationId: request.operationId,
|
|
3145
|
+
modelId: request.modelId,
|
|
3146
|
+
source: resources.find((resource) => resource.schemaVersion === "openpond.learningSource.v1"),
|
|
3147
|
+
evidence: resources.filter((resource) => resource.schemaVersion === "openpond.taskEvidence.v1"),
|
|
3148
|
+
proposals: resources.filter((resource) => resource.schemaVersion === "openpond.taskFeedbackRecord.v1")
|
|
3149
|
+
});
|
|
3150
|
+
}
|
|
2867
3151
|
export {
|
|
2868
3152
|
MAX_TASKSET_PACKAGE_BYTES,
|
|
3153
|
+
LearningDomainError3 as ModelBatchReviewError,
|
|
3154
|
+
ModelBatchReviewInspectionSchema,
|
|
3155
|
+
ModelBatchReviewReceiptSchema,
|
|
3156
|
+
ModelBatchReviewRequestSchema,
|
|
2869
3157
|
OpenPondTasksetPackageClient,
|
|
2870
3158
|
OpenPondTasksetPackageError,
|
|
2871
3159
|
TasksetPackageContentSchema,
|
|
@@ -2876,8 +3164,11 @@ export {
|
|
|
2876
3164
|
TasksetPackageReadbackSchema,
|
|
2877
3165
|
TasksetPackageReceiptSchema,
|
|
2878
3166
|
TasksetPackageSchema,
|
|
3167
|
+
beginModelBatchReview,
|
|
2879
3168
|
createTasksetPackage,
|
|
2880
3169
|
decodeTasksetPackageFile,
|
|
3170
|
+
findModelBatchReview,
|
|
3171
|
+
inspectModelBatchPackage,
|
|
2881
3172
|
learningPackageContextFiles,
|
|
2882
3173
|
tasksetPackageRewardBinding,
|
|
2883
3174
|
validateTasksetLearningResources,
|