openpond 0.0.43 → 0.0.45
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/dist/chunks/{app-layer-UPNMLDO5.js → app-layer-4PR4BS7N.js} +2 -2
- package/dist/chunks/{app-server-runtime-CLJWPISO.js → app-server-runtime-P7IXDS5G.js} +7 -7
- package/dist/chunks/{apps-XLPSFCW4.js → apps-V7TF6RTV.js} +2 -2
- package/dist/chunks/{chunk-YYLUURQ4.js → chunk-2FZEH7JB.js} +3 -3
- package/dist/chunks/{chunk-L7SIUJEV.js → chunk-5T5VWUX6.js} +1 -1
- package/dist/chunks/{chunk-2SF6HLT7.js → chunk-CGVG46T5.js} +4 -1
- package/dist/chunks/{chunk-KSOI7GFA.js → chunk-HGJQUQAP.js} +1 -1
- package/dist/chunks/{chunk-M2NXTLH2.js → chunk-I4CQJSZK.js} +518 -26
- package/dist/chunks/{chunk-ZTE6A2VE.js → chunk-IM67TWCC.js} +1 -1
- package/dist/chunks/{chunk-DBHT62E5.js → chunk-JL32HPXH.js} +1 -1
- package/dist/chunks/{chunk-SBVJGA3Y.js → chunk-RHERAMEA.js} +768 -44
- package/dist/chunks/{chunk-YEKQWRBS.js → chunk-SLTKA5AW.js} +9 -5
- package/dist/chunks/{chunk-QHUJ2XVJ.js → chunk-YBQTQWOY.js} +32 -32
- package/dist/chunks/{chunk-MCBAI5M5.js → chunk-YYJH2WWX.js} +1 -1
- package/dist/chunks/{cli-UU4WCREM.js → cli-ULGWD6S2.js} +6 -6
- package/dist/chunks/{core-commands-D66KZEHU.js → core-commands-O76V62SB.js} +3 -3
- package/dist/chunks/{desktop-test-FDA6MV77.js → desktop-test-N2UPF5PP.js} +2 -2
- package/dist/chunks/{extension-GDRW22N3.js → extension-XQBSBEVO.js} +4 -1
- package/dist/chunks/{help-7QG4UITV.js → help-YSGYNWPJ.js} +1 -1
- package/dist/chunks/{opchat-I2WGEFGU.js → opchat-F5OFTH56.js} +2 -2
- package/dist/chunks/{organizations-MADRLI2I.js → organizations-EEQPOIAZ.js} +4 -4
- package/dist/chunks/{profile-B4A3GT3U.js → profile-3OGGNTDS.js} +2 -2
- package/dist/chunks/{project-agent-PCO42FZJ.js → project-agent-LVOSTVE5.js} +2 -2
- package/dist/chunks/{sandbox-command-LBKSIR7P.js → sandbox-command-S3VMWGMH.js} +2 -2
- package/dist/chunks/{sandbox-template-YJXGG4H3.js → sandbox-template-UDXW7LRZ.js} +3 -3
- package/dist/chunks/{src-L2H4X4IY.js → src-C2CPWIJG.js} +1393 -487
- package/dist/chunks/{src-35JYS27L.js → src-JMDVVYFD.js} +3 -3
- package/dist/chunks/{teams-bot-L4SR2SWM.js → teams-bot-YRKKQDYL.js} +2 -2
- package/dist/chunks/{workspaces-ODFJT6HJ.js → workspaces-BFMHQF7F.js} +2 -2
- package/dist/cli.js +13 -13
- package/dist/web/assets/{AppDialog-Bua3f-6v.js → AppDialog-D8HiG2vU.js} +1 -1
- package/dist/web/assets/{AppsView-DrJkHeSO.js → AppsView-gZoE9vMK.js} +1 -1
- package/dist/web/assets/{BrowserSidebar-Bf-4Vozm.js → BrowserSidebar-D-HKuusA.js} +1 -1
- package/dist/web/assets/{CommandMenu-DDZlGQiB.js → CommandMenu-CzrixKzR.js} +1 -1
- package/dist/web/assets/{CommunityView-BtcCrba7.js → CommunityView-khbmjhWm.js} +1 -1
- package/dist/web/assets/{ComposerCreateImproveStrip-DOe5Otg6.js → ComposerCreateImproveStrip-CLxAZScx.js} +1 -1
- package/dist/web/assets/{GetStartedView-CG4z0Dtw.js → GetStartedView-C3nu80E_.js} +1 -1
- package/dist/web/assets/{LabModelVersionDetailPage-BEkmSKuv.js → LabModelVersionDetailPage-EcvM9TTz.js} +1 -1
- package/dist/web/assets/{LabSkillSidebar-BIotN_RB.js → LabSkillSidebar-LG9zm1yx.js} +1 -1
- package/dist/web/assets/{LabsRoute-CfOe0Zt5.js → LabsRoute-B4lijQB6.js} +3 -3
- package/dist/web/assets/{MainChatThread-C0-l1mHt.js → MainChatThread-D_-zWAT8.js} +2 -2
- package/dist/web/assets/{MainPane-aOVlVyRK.js → MainPane-Dfvc7xdq.js} +3 -3
- package/dist/web/assets/{MarkdownText-D7DT1k1k.js → MarkdownText-C0_D5zMW.js} +1 -1
- package/dist/web/assets/{Messages-CBIXruA9.js → Messages-C-Q91wIU.js} +1 -1
- package/dist/web/assets/{NativeSkillSidebar-C4t69Y7B.js → NativeSkillSidebar-DoSiN-Mr.js} +1 -1
- package/dist/web/assets/{NewProjectDialog-Bat65tYD.js → NewProjectDialog-C0hUcNck.js} +1 -1
- package/dist/web/assets/{OutputsPage-C58n39Db.js → OutputsPage-PV7Pm_OU.js} +1 -1
- package/dist/web/assets/{RightChatPanelStack-Bp7nZxQ0.js → RightChatPanelStack-C1LZTXEY.js} +1 -1
- package/dist/web/assets/{ScheduledWorkPage-BXP0j_c_.js → ScheduledWorkPage-7UGpZ8Gc.js} +1 -1
- package/dist/web/assets/{SettingsView-CtWK_aQ0.css → SettingsView-CrXwXjZ1.css} +1 -1
- package/dist/web/assets/SettingsView-DjSbBHb7.js +6 -0
- package/dist/web/assets/{TeamChatView-De2wdqyY.js → TeamChatView-D7_mZ6eT.js} +1 -1
- package/dist/web/assets/{TerminalOverlay-BsuCH3NX.js → TerminalOverlay-Cji37VqF.js} +1 -1
- package/dist/web/assets/{TrainingCreationPanel-D2OFWyQ-.js → TrainingCreationPanel-VctKu0b5.js} +1 -1
- package/dist/web/assets/{TrainingDraftPanel-DBpulXTz.js → TrainingDraftPanel-BJXcYQt5.js} +1 -1
- package/dist/web/assets/{UsageSettingsSection-DzMLUXAO.js → UsageSettingsSection-BxnbRELL.js} +1 -1
- package/dist/web/assets/{WorkspaceDiffPanel-B7yIhjNF.js → WorkspaceDiffPanel-Dj50bA58.js} +3 -3
- package/dist/web/assets/{WorkspaceEnvironmentMenu-B6jXM9xG.js → WorkspaceEnvironmentMenu-Dv_pK7bp.js} +1 -1
- package/dist/web/assets/{WorkspaceGitDialogs-CQ0fl9Qk.js → WorkspaceGitDialogs-Bs55QQk1.js} +1 -1
- package/dist/web/assets/{WorkspaceMonacoEditor-C2ctzITz.js → WorkspaceMonacoEditor-BKyihWp7.js} +3 -3
- package/dist/web/assets/{arrow-up-right-ecwcx3xt.js → arrow-up-right-O7S09juy.js} +1 -1
- package/dist/web/assets/{chevron-up-DP8PPfWd.js → chevron-up-ILypNpT9.js} +1 -1
- package/dist/web/assets/{circle-alert-envg1D6O.js → circle-alert-CqcEVbV4.js} +1 -1
- package/dist/web/assets/{cloud-upload-5MGPLnaA.js → cloud-upload-B-KanLwx.js} +1 -1
- package/dist/web/assets/{cssMode-CFifcEfw.js → cssMode-BIJYgmZR.js} +1 -1
- package/dist/web/assets/{folder-open-Dq3ZjeEP.js → folder-open-DCkv77b6.js} +1 -1
- package/dist/web/assets/{folder-plus-EiZ5wbv_.js → folder-plus-BqZUZnCI.js} +1 -1
- package/dist/web/assets/{git-branch-w29jmQ-X.js → git-branch-DLlfIOn_.js} +1 -1
- package/dist/web/assets/{git-commit-horizontal-CGZM0PZo.js → git-commit-horizontal-ClqcnbGe.js} +1 -1
- package/dist/web/assets/{htmlMode-Ct1P-_lf.js → htmlMode-B1PeNAJB.js} +1 -1
- package/dist/web/assets/index-BLgZVGpl.js +178 -0
- package/dist/web/assets/index-vV2sAtZG.js +1 -0
- package/dist/web/assets/{info-Cd2HkIeb.js → info-B-IcZCF8.js} +1 -1
- package/dist/web/assets/{jsonMode-BAUNCNAX.js → jsonMode-NqLd9KdG.js} +1 -1
- package/dist/web/assets/{lspLanguageFeatures-fucFYI1A.js → lspLanguageFeatures-DWa2E295.js} +1 -1
- package/dist/web/assets/{monaco.contribution-nG-5Cyjz.js → monaco.contribution-BYgdD9ap.js} +2 -2
- package/dist/web/assets/{monaco.contribution-DLLMi4cB.js → monaco.contribution-DOdpxIBx.js} +2 -2
- package/dist/web/assets/{monaco.contribution-g9QiIm1d.js → monaco.contribution-EiA1XLoV.js} +2 -2
- package/dist/web/assets/{monaco.contribution-Ce4v5kei.js → monaco.contribution-rE0pudHI.js} +2 -2
- package/dist/web/assets/{monitor-DgHCnCKl.js → monitor-IKHoaZYu.js} +1 -1
- package/dist/web/assets/{play-Be0XDHBu.js → play-B8a2D-Pu.js} +1 -1
- package/dist/web/assets/{python-CEhEiPo7.js → python-DXSUy3MY.js} +1 -1
- package/dist/web/assets/{refresh-cw-DH0KLbdD.js → refresh-cw-De3V2Tzv.js} +1 -1
- package/dist/web/assets/{reply-D1TK2yV8.js → reply-DtzLGSWr.js} +1 -1
- package/dist/web/assets/{save-C1MSeBnQ.js → save-CKSY2EC7.js} +1 -1
- package/dist/web/assets/{square-CemI-0g5.js → square-U2IL0jpg.js} +1 -1
- package/dist/web/assets/{square-pen-CqfsvR_n.js → square-pen-DhLj2CtT.js} +1 -1
- package/dist/web/assets/{toggleHighContrast-DrdQN3Rl.js → toggleHighContrast-Dusp2QyF.js} +1 -1
- package/dist/web/assets/{tsMode-CKB5oOUU.js → tsMode-CarGQUBk.js} +1 -1
- package/dist/web/assets/{upload-t751LLqP.js → upload-12FovSVQ.js} +1 -1
- package/dist/web/assets/{useLocalAgentSchedules-CzC3wfIP.js → useLocalAgentSchedules-DGQQtn8W.js} +1 -1
- package/dist/web/assets/{wifi-off-CxTHH3JU.js → wifi-off-CjQEPd2q.js} +1 -1
- package/dist/web/assets/{workers-CqYddzCd.js → workers-CeBkFaBO.js} +1 -1
- package/dist/web/assets/{yaml-e5wBkc-C.js → yaml-DQPBhykU.js} +1 -1
- package/dist/web/index.html +1 -1
- package/package.json +1 -1
- package/dist/web/assets/SettingsView-DOOfhO9a.js +0 -6
- package/dist/web/assets/index-ByIQyiUl.js +0 -178
- package/dist/web/assets/index-D4uCeiep.js +0 -1
|
@@ -844,12 +844,12 @@ var SandboxTemplateRuntimeSchema = external_exports.object({
|
|
|
844
844
|
snapshot: external_exports.string().trim().min(1).max(191).optional(),
|
|
845
845
|
image: DockerfileRuntimeImageSchema.optional(),
|
|
846
846
|
dockerfile: DockerfileRuntimeSourceSchema.optional()
|
|
847
|
-
}).strict().superRefine((
|
|
847
|
+
}).strict().superRefine((runtime2, context) => {
|
|
848
848
|
const selected = [
|
|
849
|
-
|
|
850
|
-
|
|
851
|
-
|
|
852
|
-
|
|
849
|
+
runtime2.base,
|
|
850
|
+
runtime2.snapshot,
|
|
851
|
+
runtime2.image,
|
|
852
|
+
runtime2.dockerfile
|
|
853
853
|
].filter(Boolean).length;
|
|
854
854
|
if (selected !== 1) {
|
|
855
855
|
context.addIssue({
|
|
@@ -1803,6 +1803,195 @@ function createHarnessAdvanceReceipt(content) {
|
|
|
1803
1803
|
});
|
|
1804
1804
|
}
|
|
1805
1805
|
|
|
1806
|
+
// ../../packages/harness/dist/evaluation-review.js
|
|
1807
|
+
var BoundedTextSchema2 = external_exports.string().trim().min(1).max(1e5);
|
|
1808
|
+
var HarnessEvaluationReviewClassificationSchema = external_exports.enum([
|
|
1809
|
+
"no_action",
|
|
1810
|
+
"harness_maintenance",
|
|
1811
|
+
"runtime",
|
|
1812
|
+
"product",
|
|
1813
|
+
"taskset",
|
|
1814
|
+
"model_improvement"
|
|
1815
|
+
]);
|
|
1816
|
+
var HarnessEvaluationReviewAuthoritySchema = external_exports.enum([
|
|
1817
|
+
"none",
|
|
1818
|
+
"runtime_service",
|
|
1819
|
+
"product_team",
|
|
1820
|
+
"human_review",
|
|
1821
|
+
"evaluation_system",
|
|
1822
|
+
"training_system"
|
|
1823
|
+
]);
|
|
1824
|
+
var HarnessReviewEvidenceKindSchema = external_exports.enum([
|
|
1825
|
+
"observation",
|
|
1826
|
+
"trigger",
|
|
1827
|
+
"route_decision",
|
|
1828
|
+
"refiner_outcome",
|
|
1829
|
+
"proposal",
|
|
1830
|
+
"validation",
|
|
1831
|
+
"apply_receipt",
|
|
1832
|
+
"harness_advance",
|
|
1833
|
+
"rollback",
|
|
1834
|
+
"work_outcome",
|
|
1835
|
+
"taskset",
|
|
1836
|
+
"evaluation",
|
|
1837
|
+
"training_qualification",
|
|
1838
|
+
"model_candidate"
|
|
1839
|
+
]);
|
|
1840
|
+
var HarnessReviewOwnerScopeSchema = external_exports.object({
|
|
1841
|
+
kind: external_exports.enum(["personal", "team"]),
|
|
1842
|
+
id: ReleaseIdSchema
|
|
1843
|
+
}).strict();
|
|
1844
|
+
var HarnessReviewSourcePolicyRefSchema = external_exports.object({
|
|
1845
|
+
policy: ImmutableReleaseRefSchema,
|
|
1846
|
+
state: external_exports.enum(["authorized", "revoked", "deleted", "expired"]),
|
|
1847
|
+
checkedAt: ReleaseTimestampSchema
|
|
1848
|
+
}).strict();
|
|
1849
|
+
var HarnessReviewEvidenceRefSchema = external_exports.object({
|
|
1850
|
+
evidence: ImmutableReleaseRefSchema,
|
|
1851
|
+
kind: HarnessReviewEvidenceKindSchema,
|
|
1852
|
+
sourceRef: ReleaseIdSchema,
|
|
1853
|
+
sourcePolicy: HarnessReviewSourcePolicyRefSchema,
|
|
1854
|
+
occurrenceKey: ReleaseHashSchema,
|
|
1855
|
+
occurredAt: ReleaseTimestampSchema
|
|
1856
|
+
}).strict();
|
|
1857
|
+
var HarnessReviewExcludedEvidenceSchema = external_exports.object({
|
|
1858
|
+
evidence: ImmutableReleaseRefSchema,
|
|
1859
|
+
sourcePolicy: HarnessReviewSourcePolicyRefSchema.nullable(),
|
|
1860
|
+
reason: external_exports.enum([
|
|
1861
|
+
"outside_scope",
|
|
1862
|
+
"before_watermark",
|
|
1863
|
+
"duplicate",
|
|
1864
|
+
"resolved",
|
|
1865
|
+
"revoked",
|
|
1866
|
+
"deleted",
|
|
1867
|
+
"expired",
|
|
1868
|
+
"sensitive",
|
|
1869
|
+
"unverified",
|
|
1870
|
+
"budget"
|
|
1871
|
+
])
|
|
1872
|
+
}).strict();
|
|
1873
|
+
var HarnessReviewWatermarkSchema = external_exports.object({
|
|
1874
|
+
cursor: ReleaseHashSchema,
|
|
1875
|
+
throughCreatedAt: ReleaseTimestampSchema
|
|
1876
|
+
}).strict();
|
|
1877
|
+
var HarnessReviewClaimSchema = external_exports.object({
|
|
1878
|
+
fingerprint: ReleaseHashSchema,
|
|
1879
|
+
recurrenceFamily: external_exports.string().trim().min(1).max(1e3),
|
|
1880
|
+
statement: BoundedTextSchema2,
|
|
1881
|
+
independentOccurrences: external_exports.number().int().positive().max(1e6),
|
|
1882
|
+
unresolvedOccurrences: external_exports.number().int().positive().max(1e6)
|
|
1883
|
+
}).strict().refine((claim2) => claim2.unresolvedOccurrences <= claim2.independentOccurrences, "unresolved occurrences cannot exceed independent occurrences");
|
|
1884
|
+
var HarnessReviewTriageLayerSchema = external_exports.enum([
|
|
1885
|
+
"harness",
|
|
1886
|
+
"runtime",
|
|
1887
|
+
"product",
|
|
1888
|
+
"retrieval",
|
|
1889
|
+
"tools",
|
|
1890
|
+
"evaluation",
|
|
1891
|
+
"model"
|
|
1892
|
+
]);
|
|
1893
|
+
var HarnessReviewTriageDecisionSchema = external_exports.object({
|
|
1894
|
+
layer: HarnessReviewTriageLayerSchema,
|
|
1895
|
+
status: external_exports.enum(["not_applicable", "unresolved", "resolved", "blocked"]),
|
|
1896
|
+
reason: BoundedTextSchema2,
|
|
1897
|
+
evidenceRefs: external_exports.array(ImmutableReleaseRefSchema).max(1e3)
|
|
1898
|
+
}).strict();
|
|
1899
|
+
var HarnessEvaluationReviewReceiptContentSchema = external_exports.object({
|
|
1900
|
+
schemaVersion: external_exports.literal("openpond.harnessEvaluationReviewReceipt.v1"),
|
|
1901
|
+
id: ReleaseIdSchema,
|
|
1902
|
+
ownerScope: HarnessReviewOwnerScopeSchema,
|
|
1903
|
+
workspaceRef: ReleaseIdSchema,
|
|
1904
|
+
harnessRelease: ImmutableReleaseRefSchema,
|
|
1905
|
+
previousWatermark: HarnessReviewWatermarkSchema.nullable(),
|
|
1906
|
+
nextWatermark: HarnessReviewWatermarkSchema,
|
|
1907
|
+
selectedEvidence: external_exports.array(HarnessReviewEvidenceRefSchema).max(1e4),
|
|
1908
|
+
excludedEvidence: external_exports.array(HarnessReviewExcludedEvidenceSchema).max(1e4),
|
|
1909
|
+
claim: HarnessReviewClaimSchema.nullable(),
|
|
1910
|
+
classification: HarnessEvaluationReviewClassificationSchema,
|
|
1911
|
+
triage: external_exports.array(HarnessReviewTriageDecisionSchema).max(16),
|
|
1912
|
+
reason: BoundedTextSchema2,
|
|
1913
|
+
nextAuthority: HarnessEvaluationReviewAuthoritySchema,
|
|
1914
|
+
maxEstimatedCostUsd: external_exports.number().finite().nonnegative(),
|
|
1915
|
+
tasksetProposal: ImmutableReleaseRefSchema.nullable(),
|
|
1916
|
+
evaluation: ImmutableReleaseRefSchema.nullable(),
|
|
1917
|
+
trainingQualification: ImmutableReleaseRefSchema.nullable(),
|
|
1918
|
+
policyVersion: ReleaseIdSchema,
|
|
1919
|
+
createdAt: ReleaseTimestampSchema,
|
|
1920
|
+
metadata: MetadataSchema
|
|
1921
|
+
}).strict().superRefine((receipt, context) => {
|
|
1922
|
+
const selectedKeys = receipt.selectedEvidence.map((item) => `${item.evidence.id}:${item.evidence.contentHash}`);
|
|
1923
|
+
if (new Set(selectedKeys).size !== selectedKeys.length) {
|
|
1924
|
+
context.addIssue({
|
|
1925
|
+
code: "custom",
|
|
1926
|
+
message: "selected review evidence must be unique",
|
|
1927
|
+
path: ["selectedEvidence"]
|
|
1928
|
+
});
|
|
1929
|
+
}
|
|
1930
|
+
if (receipt.selectedEvidence.some((item) => item.sourcePolicy.state !== "authorized")) {
|
|
1931
|
+
context.addIssue({
|
|
1932
|
+
code: "custom",
|
|
1933
|
+
message: "selected review evidence must be authorized at review time",
|
|
1934
|
+
path: ["selectedEvidence"]
|
|
1935
|
+
});
|
|
1936
|
+
}
|
|
1937
|
+
if (receipt.classification === "no_action") {
|
|
1938
|
+
if (receipt.nextAuthority !== "none") {
|
|
1939
|
+
context.addIssue({
|
|
1940
|
+
code: "custom",
|
|
1941
|
+
message: "no-action review receipts require no next authority",
|
|
1942
|
+
path: ["nextAuthority"]
|
|
1943
|
+
});
|
|
1944
|
+
}
|
|
1945
|
+
if (receipt.tasksetProposal || receipt.evaluation || receipt.trainingQualification) {
|
|
1946
|
+
context.addIssue({
|
|
1947
|
+
code: "custom",
|
|
1948
|
+
message: "no-action review receipts cannot carry downstream refs"
|
|
1949
|
+
});
|
|
1950
|
+
}
|
|
1951
|
+
} else if (!receipt.claim || receipt.selectedEvidence.length === 0) {
|
|
1952
|
+
context.addIssue({
|
|
1953
|
+
code: "custom",
|
|
1954
|
+
message: "actionable review receipts require a claim and selected evidence"
|
|
1955
|
+
});
|
|
1956
|
+
}
|
|
1957
|
+
if (receipt.classification === "runtime" && receipt.nextAuthority !== "runtime_service") {
|
|
1958
|
+
context.addIssue({
|
|
1959
|
+
code: "custom",
|
|
1960
|
+
message: "runtime classifications route to runtime-service authority",
|
|
1961
|
+
path: ["nextAuthority"]
|
|
1962
|
+
});
|
|
1963
|
+
}
|
|
1964
|
+
if (receipt.classification === "product" && receipt.nextAuthority !== "product_team") {
|
|
1965
|
+
context.addIssue({
|
|
1966
|
+
code: "custom",
|
|
1967
|
+
message: "product classifications route to product-team authority",
|
|
1968
|
+
path: ["nextAuthority"]
|
|
1969
|
+
});
|
|
1970
|
+
}
|
|
1971
|
+
if (receipt.classification === "taskset" && receipt.nextAuthority !== "human_review") {
|
|
1972
|
+
context.addIssue({
|
|
1973
|
+
code: "custom",
|
|
1974
|
+
message: "Taskset classifications require human review"
|
|
1975
|
+
});
|
|
1976
|
+
}
|
|
1977
|
+
if (receipt.classification === "model_improvement" && (!receipt.evaluation || !receipt.trainingQualification || !["human_review", "training_system"].includes(receipt.nextAuthority))) {
|
|
1978
|
+
context.addIssue({
|
|
1979
|
+
code: "custom",
|
|
1980
|
+
message: "model-improvement classifications require Evaluation and qualification refs plus explicit authority"
|
|
1981
|
+
});
|
|
1982
|
+
}
|
|
1983
|
+
});
|
|
1984
|
+
var HarnessEvaluationReviewReceiptSchema = HarnessEvaluationReviewReceiptContentSchema.extend({
|
|
1985
|
+
contentHash: ReleaseHashSchema
|
|
1986
|
+
}).strict();
|
|
1987
|
+
function createHarnessEvaluationReviewReceipt(input) {
|
|
1988
|
+
const content = HarnessEvaluationReviewReceiptContentSchema.parse(input);
|
|
1989
|
+
return HarnessEvaluationReviewReceiptSchema.parse({
|
|
1990
|
+
...content,
|
|
1991
|
+
contentHash: contentHash(content)
|
|
1992
|
+
});
|
|
1993
|
+
}
|
|
1994
|
+
|
|
1806
1995
|
// ../../packages/harness/dist/tools.js
|
|
1807
1996
|
var ToolDeclarationSchema = external_exports.object({
|
|
1808
1997
|
name: external_exports.string().trim().min(1).max(64).regex(/^[a-zA-Z][a-zA-Z0-9_-]*$/),
|
|
@@ -1909,7 +2098,7 @@ var HarnessTraceSchema = external_exports.object({
|
|
|
1909
2098
|
}).strict();
|
|
1910
2099
|
|
|
1911
2100
|
// ../../packages/harness/dist/harness-improvements.js
|
|
1912
|
-
var
|
|
2101
|
+
var BoundedTextSchema3 = external_exports.string().trim().min(1).max(1e5);
|
|
1913
2102
|
var MetadataSchema3 = external_exports.record(external_exports.string(), external_exports.unknown()).default({});
|
|
1914
2103
|
var ImprovementSafeBoundaryKindSchema = external_exports.enum([
|
|
1915
2104
|
"completed_tool_batch",
|
|
@@ -1957,11 +2146,11 @@ var ImprovementObservationContentSchema = external_exports.object({
|
|
|
1957
2146
|
state: ImprovementObservationStateSchema,
|
|
1958
2147
|
tool: ImprovementToolIdentitySchema.nullable(),
|
|
1959
2148
|
deterministicClass: external_exports.string().trim().min(1).max(500).nullable(),
|
|
1960
|
-
summary:
|
|
2149
|
+
summary: BoundedTextSchema3,
|
|
1961
2150
|
createdAt: ReleaseTimestampSchema,
|
|
1962
2151
|
metadata: MetadataSchema3
|
|
1963
2152
|
}).strict().superRefine((observation, context) => {
|
|
1964
|
-
if (new Set(observation.eventRefs.map((
|
|
2153
|
+
if (new Set(observation.eventRefs.map((reference2) => reference2.id)).size !== observation.eventRefs.length) {
|
|
1965
2154
|
context.addIssue({
|
|
1966
2155
|
code: "custom",
|
|
1967
2156
|
message: "observation event refs must be unique",
|
|
@@ -2004,7 +2193,7 @@ var RefinementTriggerDecisionContentSchema = external_exports.object({
|
|
|
2004
2193
|
decision: RefinementTriggerDecisionKindSchema,
|
|
2005
2194
|
deterministicRoute: HarnessImprovementRouteSchema.nullable(),
|
|
2006
2195
|
suggestedRoutes: external_exports.array(HarnessImprovementRouteSchema).max(8),
|
|
2007
|
-
reason:
|
|
2196
|
+
reason: BoundedTextSchema3,
|
|
2008
2197
|
deduplicationKey: ReleaseHashSchema,
|
|
2009
2198
|
policy: RefinementTriggerPolicySchema,
|
|
2010
2199
|
estimatedMaxCostUsd: external_exports.number().finite().nonnegative(),
|
|
@@ -2081,7 +2270,7 @@ var ImprovementRouteDecisionContentSchema = external_exports.object({
|
|
|
2081
2270
|
route: HarnessImprovementRouteSchema,
|
|
2082
2271
|
authority: ImprovementRouteAuthoritySchema,
|
|
2083
2272
|
automatic: external_exports.boolean(),
|
|
2084
|
-
reason:
|
|
2273
|
+
reason: BoundedTextSchema3,
|
|
2085
2274
|
createdAt: ReleaseTimestampSchema,
|
|
2086
2275
|
metadata: MetadataSchema3
|
|
2087
2276
|
}).strict().superRefine((decision2, context) => {
|
|
@@ -2109,7 +2298,7 @@ var HarnessRefinerOutcomeContentSchema = external_exports.object({
|
|
|
2109
2298
|
trigger: ImmutableReleaseRefSchema,
|
|
2110
2299
|
decision: external_exports.enum(["no_action", "proposed"]),
|
|
2111
2300
|
proposal: ImmutableReleaseRefSchema.nullable(),
|
|
2112
|
-
reason:
|
|
2301
|
+
reason: BoundedTextSchema3,
|
|
2113
2302
|
evidenceRefs: external_exports.array(ImmutableReleaseRefSchema).max(100),
|
|
2114
2303
|
estimatedCostUsd: external_exports.number().finite().nonnegative(),
|
|
2115
2304
|
createdAt: ReleaseTimestampSchema,
|
|
@@ -2301,7 +2490,7 @@ async function authorLocalHarnessRefinementWithModel(input) {
|
|
|
2301
2490
|
timeout.cleanup();
|
|
2302
2491
|
}
|
|
2303
2492
|
}
|
|
2304
|
-
function refinerMessages(
|
|
2493
|
+
function refinerMessages(evidence2) {
|
|
2305
2494
|
return [
|
|
2306
2495
|
{
|
|
2307
2496
|
role: "system",
|
|
@@ -2313,6 +2502,8 @@ function refinerMessages(evidence) {
|
|
|
2313
2502
|
"Choose the smallest correct route: runtime for dependency/tool/capability defects; memory for durable facts or preferences; prompt for broad behavioral guidance; skill for a repeatable workflow or tool strategy; agent for a reusable role; product for application defects; taskset for a controlled behavioral measurement need; training only for a persistent model-policy gap; no_action for one-off or low-value evidence.",
|
|
2314
2503
|
"Distinguish a broken required runtime from a bad tool strategy. Route to runtime when the supported dependency or capability itself is missing or broken. Propose a Skill when the successful recovery proves an already-supported path that future agents should select before an unavailable or wasteful alternative.",
|
|
2315
2504
|
"Use decision=route for runtime, product, taskset, or training. These routes create an inspectable recommendation and never mutate the Harness in this Refiner step.",
|
|
2505
|
+
"A completed refine_request tool call only requests this bounded review; it does not create or complete a route by itself. Never return no_action merely because refine_request completed, supplied a suggested route, or because a separate system owns the routed work. Evaluate the evidence, and when it supports that external route, return decision=route so the immutable handoff receipt is actually recorded.",
|
|
2506
|
+
"For training, the Refiner records one occurrence; downstream receipt review owns recurrence thresholds, Taskset creation, qualification, approval, and training. Do not require those downstream steps before recording a grounded persistent model-policy occurrence, and do not imply that decision=route starts or binds training.",
|
|
2316
2507
|
"Use decision=propose for memory, prompt, skill, or agent component CRUD. Memory is externally stored bounded context, not a Harness source file.",
|
|
2317
2508
|
"Desktop Work currently activates released instructions and standalone Skills, but it has no Agent source compiler or executor. Do not propose Agent create/update changes for this runtime. Use prompt or Skill when that is the smallest active component, or route to product when an Agent executor is actually required.",
|
|
2318
2509
|
"An update/delete target must match sourceCatalog and its route kind. Never update an unrelated component merely because it is available.",
|
|
@@ -2337,7 +2528,7 @@ function refinerMessages(evidence) {
|
|
|
2337
2528
|
},
|
|
2338
2529
|
{
|
|
2339
2530
|
role: "user",
|
|
2340
|
-
content: JSON.stringify(
|
|
2531
|
+
content: JSON.stringify(evidence2, null, 2)
|
|
2341
2532
|
}
|
|
2342
2533
|
];
|
|
2343
2534
|
}
|
|
@@ -2653,7 +2844,7 @@ function collectObservations(input) {
|
|
|
2653
2844
|
});
|
|
2654
2845
|
if (recoveredFailureKeys.has(failureKey))
|
|
2655
2846
|
continue;
|
|
2656
|
-
const failureObservationIndex = observations.findIndex((observation) => observation.kind === "tool_failure" && observation.eventRefs.some((
|
|
2847
|
+
const failureObservationIndex = observations.findIndex((observation) => observation.kind === "tool_failure" && observation.eventRefs.some((reference2) => reference2.id === priorFailure.event.id));
|
|
2657
2848
|
if (failureObservationIndex >= 0) {
|
|
2658
2849
|
observations[failureObservationIndex] = observationFor({
|
|
2659
2850
|
input,
|
|
@@ -2687,7 +2878,7 @@ function collectObservations(input) {
|
|
|
2687
2878
|
}
|
|
2688
2879
|
const recovered = observations.filter((observation) => observation.kind === "recovery");
|
|
2689
2880
|
if (input.boundary.kind === "turn_completed" && recovered.length > 0) {
|
|
2690
|
-
const relevantOutcomes = input.outcomes.filter((outcome) => recovered.some((observation) => observation.eventRefs.some((
|
|
2881
|
+
const relevantOutcomes = input.outcomes.filter((outcome) => recovered.some((observation) => observation.eventRefs.some((reference2) => reference2.id === outcome.event.id)));
|
|
2691
2882
|
observations.push(observationFor({
|
|
2692
2883
|
input,
|
|
2693
2884
|
kind: "completion_detour",
|
|
@@ -2968,8 +3159,8 @@ function proposalEvidence(observations) {
|
|
|
2968
3159
|
const byId = /* @__PURE__ */ new Map();
|
|
2969
3160
|
for (const observation of observations) {
|
|
2970
3161
|
for (const event of observation.eventRefs) {
|
|
2971
|
-
const
|
|
2972
|
-
byId.set(`${
|
|
3162
|
+
const evidence2 = observationEvidence(observation, event);
|
|
3163
|
+
byId.set(`${evidence2.kind}:${evidence2.id}`, evidence2);
|
|
2973
3164
|
}
|
|
2974
3165
|
}
|
|
2975
3166
|
return [...byId.values()];
|
|
@@ -2990,8 +3181,8 @@ function uniqueEventRefs(observations) {
|
|
|
2990
3181
|
}
|
|
2991
3182
|
return [...byId.values()];
|
|
2992
3183
|
}
|
|
2993
|
-
function sameOverlayRef(overlay,
|
|
2994
|
-
return overlay.id ===
|
|
3184
|
+
function sameOverlayRef(overlay, reference2) {
|
|
3185
|
+
return overlay.id === reference2.id && overlay.revision === reference2.revision && overlay.contentHash === reference2.contentHash;
|
|
2995
3186
|
}
|
|
2996
3187
|
function overlayRef(overlay) {
|
|
2997
3188
|
return {
|
|
@@ -3564,11 +3755,11 @@ function createWorkEvidenceReceipt(input) {
|
|
|
3564
3755
|
const receipt = WorkEvidenceReceiptContentSchema.parse(input);
|
|
3565
3756
|
return WorkEvidenceReceiptSchema.parse({ ...receipt, contentHash: contentHash(receipt) });
|
|
3566
3757
|
}
|
|
3567
|
-
function createWorkFeedbackReceipt(input,
|
|
3758
|
+
function createWorkFeedbackReceipt(input, evidence2) {
|
|
3568
3759
|
const feedback = WorkFeedbackReceiptContentSchema.parse(input);
|
|
3569
3760
|
const receipt = WorkFeedbackReceiptSchema.parse({ ...feedback, contentHash: contentHash(feedback) });
|
|
3570
|
-
if (
|
|
3571
|
-
assertFeedbackTargetsEvidence(receipt,
|
|
3761
|
+
if (evidence2)
|
|
3762
|
+
assertFeedbackTargetsEvidence(receipt, evidence2);
|
|
3572
3763
|
return receipt;
|
|
3573
3764
|
}
|
|
3574
3765
|
function verifyWorkEvidenceReceipt(value) {
|
|
@@ -3581,11 +3772,11 @@ function workEvidenceReceiptRef(receipt) {
|
|
|
3581
3772
|
sizeBytes: null
|
|
3582
3773
|
});
|
|
3583
3774
|
}
|
|
3584
|
-
function assertFeedbackTargetsEvidence(feedback,
|
|
3585
|
-
if (feedback.evidenceReceiptRef.contentHash !==
|
|
3775
|
+
function assertFeedbackTargetsEvidence(feedback, evidence2) {
|
|
3776
|
+
if (feedback.evidenceReceiptRef.contentHash !== evidence2.contentHash) {
|
|
3586
3777
|
throw new Error("Feedback targets a different Work evidence receipt.");
|
|
3587
3778
|
}
|
|
3588
|
-
if (feedback.outputRevisionRef && !
|
|
3779
|
+
if (feedback.outputRevisionRef && !evidence2.outputRefs.some((output2) => output2.contentHash === feedback.outputRevisionRef.contentHash)) {
|
|
3589
3780
|
throw new Error("Feedback targets an output revision not bound by the Work evidence receipt.");
|
|
3590
3781
|
}
|
|
3591
3782
|
}
|
|
@@ -3959,6 +4150,300 @@ var GraderEvidenceContentSchema = external_exports.object({
|
|
|
3959
4150
|
}).strict();
|
|
3960
4151
|
var GraderEvidenceSchema = GraderEvidenceContentSchema.extend({ contentHash: ReleaseHashSchema }).strict();
|
|
3961
4152
|
|
|
4153
|
+
// ../../packages/evals/dist/model-improvement-qualification.js
|
|
4154
|
+
var BoundedTextSchema4 = external_exports.string().trim().min(1).max(1e5);
|
|
4155
|
+
var ModelImprovementDecisionSchema = external_exports.enum([
|
|
4156
|
+
"no_training",
|
|
4157
|
+
"sft",
|
|
4158
|
+
"preference",
|
|
4159
|
+
"rl"
|
|
4160
|
+
]);
|
|
4161
|
+
var ModelImprovementSignalSchema = external_exports.object({
|
|
4162
|
+
kind: external_exports.enum(["none", "demonstrations", "chosen_rejected", "scalar_reward"]),
|
|
4163
|
+
strength: external_exports.enum(["absent", "weak", "usable"]),
|
|
4164
|
+
calibrated: external_exports.boolean(),
|
|
4165
|
+
confounded: external_exports.boolean(),
|
|
4166
|
+
variance: external_exports.number().finite().nonnegative().nullable(),
|
|
4167
|
+
evidenceRefs: external_exports.array(ImmutableReleaseRefSchema).max(1e5)
|
|
4168
|
+
}).strict();
|
|
4169
|
+
var ModelImprovementQualificationReceiptContentSchema = external_exports.object({
|
|
4170
|
+
schemaVersion: external_exports.literal("openpond.modelImprovementQualificationReceipt.v1"),
|
|
4171
|
+
id: ReleaseIdSchema,
|
|
4172
|
+
review: ImmutableReleaseRefSchema,
|
|
4173
|
+
harnessRelease: ImmutableReleaseRefSchema,
|
|
4174
|
+
tasksetRelease: ImmutableReleaseRefSchema.nullable(),
|
|
4175
|
+
baselineEvaluation: ImmutableReleaseRefSchema.nullable(),
|
|
4176
|
+
model: ModelRefSchema,
|
|
4177
|
+
environmentHash: ReleaseHashSchema.nullable(),
|
|
4178
|
+
toolContractHash: ReleaseHashSchema.nullable(),
|
|
4179
|
+
permissionContractHash: ReleaseHashSchema.nullable(),
|
|
4180
|
+
policyHash: ReleaseHashSchema.nullable(),
|
|
4181
|
+
verifierRef: ImmutableReleaseRefSchema.nullable(),
|
|
4182
|
+
sourcePolicies: external_exports.array(HarnessReviewSourcePolicyRefSchema).max(1e4),
|
|
4183
|
+
trainingEvidenceRefs: external_exports.array(ImmutableReleaseRefSchema).max(1e5),
|
|
4184
|
+
frozenEvaluationEvidenceRefs: external_exports.array(ImmutableReleaseRefSchema).max(1e5),
|
|
4185
|
+
privacyApproval: ImmutableReleaseRefSchema.nullable(),
|
|
4186
|
+
budgetApproval: ImmutableReleaseRefSchema.nullable(),
|
|
4187
|
+
maximumCostUsd: external_exports.number().finite().nonnegative(),
|
|
4188
|
+
signal: ModelImprovementSignalSchema,
|
|
4189
|
+
decision: ModelImprovementDecisionSchema,
|
|
4190
|
+
reasons: external_exports.array(BoundedTextSchema4).min(1).max(100),
|
|
4191
|
+
createdAt: ReleaseTimestampSchema,
|
|
4192
|
+
metadata: MetadataSchema
|
|
4193
|
+
}).strict().superRefine((receipt, context) => {
|
|
4194
|
+
const trainingKeys = new Set(receipt.trainingEvidenceRefs.map(refKey));
|
|
4195
|
+
if (receipt.frozenEvaluationEvidenceRefs.some((reference2) => trainingKeys.has(refKey(reference2)))) {
|
|
4196
|
+
context.addIssue({
|
|
4197
|
+
code: "custom",
|
|
4198
|
+
message: "frozen Evaluation evidence cannot be used as training evidence",
|
|
4199
|
+
path: ["frozenEvaluationEvidenceRefs"]
|
|
4200
|
+
});
|
|
4201
|
+
}
|
|
4202
|
+
if (receipt.decision === "no_training")
|
|
4203
|
+
return;
|
|
4204
|
+
const missingGate = !receipt.tasksetRelease || !receipt.baselineEvaluation || !receipt.environmentHash || !receipt.toolContractHash || !receipt.permissionContractHash || !receipt.policyHash || !receipt.verifierRef || !receipt.privacyApproval || !receipt.budgetApproval || receipt.sourcePolicies.length === 0 || receipt.sourcePolicies.some((policy) => policy.state !== "authorized") || receipt.trainingEvidenceRefs.length === 0 || receipt.signal.strength !== "usable" || !receipt.signal.calibrated || receipt.signal.confounded;
|
|
4205
|
+
if (missingGate) {
|
|
4206
|
+
context.addIssue({
|
|
4207
|
+
code: "custom",
|
|
4208
|
+
message: "qualified model improvement requires frozen lineage, authorized signal, privacy, and budget gates"
|
|
4209
|
+
});
|
|
4210
|
+
}
|
|
4211
|
+
if (receipt.decision === "sft" && receipt.signal.kind !== "demonstrations") {
|
|
4212
|
+
context.addIssue({
|
|
4213
|
+
code: "custom",
|
|
4214
|
+
message: "SFT qualification requires demonstration signal",
|
|
4215
|
+
path: ["signal", "kind"]
|
|
4216
|
+
});
|
|
4217
|
+
}
|
|
4218
|
+
if (receipt.decision === "preference" && receipt.signal.kind !== "chosen_rejected") {
|
|
4219
|
+
context.addIssue({
|
|
4220
|
+
code: "custom",
|
|
4221
|
+
message: "preference qualification requires chosen/rejected signal",
|
|
4222
|
+
path: ["signal", "kind"]
|
|
4223
|
+
});
|
|
4224
|
+
}
|
|
4225
|
+
if (receipt.decision === "rl" && (receipt.signal.kind !== "scalar_reward" || receipt.signal.variance === null || receipt.signal.variance <= 0)) {
|
|
4226
|
+
context.addIssue({
|
|
4227
|
+
code: "custom",
|
|
4228
|
+
message: "RL qualification requires a usable scalar reward with variance",
|
|
4229
|
+
path: ["signal"]
|
|
4230
|
+
});
|
|
4231
|
+
}
|
|
4232
|
+
});
|
|
4233
|
+
var ModelImprovementQualificationReceiptSchema = ModelImprovementQualificationReceiptContentSchema.extend({
|
|
4234
|
+
contentHash: ReleaseHashSchema
|
|
4235
|
+
}).strict();
|
|
4236
|
+
function createModelImprovementQualificationReceipt(input) {
|
|
4237
|
+
const content = ModelImprovementQualificationReceiptContentSchema.parse(input);
|
|
4238
|
+
return ModelImprovementQualificationReceiptSchema.parse({
|
|
4239
|
+
...content,
|
|
4240
|
+
contentHash: contentHash(content)
|
|
4241
|
+
});
|
|
4242
|
+
}
|
|
4243
|
+
function refKey(reference2) {
|
|
4244
|
+
return `${reference2.id}:${reference2.contentHash}`;
|
|
4245
|
+
}
|
|
4246
|
+
|
|
4247
|
+
// ../../packages/evals/dist/review-conformance.js
|
|
4248
|
+
var createdAt = "2026-08-08T12:00:00.000Z";
|
|
4249
|
+
var harnessRelease = ref("harness-release");
|
|
4250
|
+
var sourcePolicy = {
|
|
4251
|
+
policy: ref("source-policy"),
|
|
4252
|
+
state: "authorized",
|
|
4253
|
+
checkedAt: createdAt
|
|
4254
|
+
};
|
|
4255
|
+
var evidence = (id, kind) => ({
|
|
4256
|
+
evidence: ref(id),
|
|
4257
|
+
kind,
|
|
4258
|
+
sourceRef: `source-${id}`,
|
|
4259
|
+
sourcePolicy,
|
|
4260
|
+
occurrenceKey: contentHash(`occurrence-${id}`),
|
|
4261
|
+
occurredAt: createdAt
|
|
4262
|
+
});
|
|
4263
|
+
var watermark = {
|
|
4264
|
+
cursor: contentHash("review-watermark"),
|
|
4265
|
+
throughCreatedAt: createdAt
|
|
4266
|
+
};
|
|
4267
|
+
var claim = (family, count = 3) => ({
|
|
4268
|
+
fingerprint: contentHash(`claim-${family}`),
|
|
4269
|
+
recurrenceFamily: family,
|
|
4270
|
+
statement: `The ${family} behavior remains unresolved after smaller-layer triage.`,
|
|
4271
|
+
independentOccurrences: count,
|
|
4272
|
+
unresolvedOccurrences: count
|
|
4273
|
+
});
|
|
4274
|
+
var noAction = createHarnessEvaluationReviewReceipt({
|
|
4275
|
+
schemaVersion: "openpond.harnessEvaluationReviewReceipt.v1",
|
|
4276
|
+
id: "review-no-action",
|
|
4277
|
+
ownerScope: { kind: "personal", id: "owner-1" },
|
|
4278
|
+
workspaceRef: "workspace-1",
|
|
4279
|
+
harnessRelease,
|
|
4280
|
+
previousWatermark: null,
|
|
4281
|
+
nextWatermark: watermark,
|
|
4282
|
+
selectedEvidence: [],
|
|
4283
|
+
excludedEvidence: [],
|
|
4284
|
+
claim: null,
|
|
4285
|
+
classification: "no_action",
|
|
4286
|
+
triage: [],
|
|
4287
|
+
reason: "The bounded evidence window contains no unresolved reusable claim.",
|
|
4288
|
+
nextAuthority: "none",
|
|
4289
|
+
maxEstimatedCostUsd: 0,
|
|
4290
|
+
tasksetProposal: null,
|
|
4291
|
+
evaluation: null,
|
|
4292
|
+
trainingQualification: null,
|
|
4293
|
+
policyVersion: "harness-review-policy-v1",
|
|
4294
|
+
createdAt,
|
|
4295
|
+
metadata: {}
|
|
4296
|
+
});
|
|
4297
|
+
var runtime = createHarnessEvaluationReviewReceipt({
|
|
4298
|
+
...withoutHash2(noAction),
|
|
4299
|
+
id: "review-runtime",
|
|
4300
|
+
selectedEvidence: [evidence("runtime-failure", "observation")],
|
|
4301
|
+
claim: claim("runtime-transport-failure", 1),
|
|
4302
|
+
classification: "runtime",
|
|
4303
|
+
triage: [
|
|
4304
|
+
{
|
|
4305
|
+
layer: "runtime",
|
|
4306
|
+
status: "unresolved",
|
|
4307
|
+
reason: "The adapter failed before model policy could affect the result.",
|
|
4308
|
+
evidenceRefs: [ref("runtime-failure")]
|
|
4309
|
+
}
|
|
4310
|
+
],
|
|
4311
|
+
reason: "A deterministic runtime regression is the smallest correct fix.",
|
|
4312
|
+
nextAuthority: "runtime_service"
|
|
4313
|
+
});
|
|
4314
|
+
var product = createHarnessEvaluationReviewReceipt({
|
|
4315
|
+
...withoutHash2(noAction),
|
|
4316
|
+
id: "review-product",
|
|
4317
|
+
selectedEvidence: [evidence("product-routing", "route_decision")],
|
|
4318
|
+
claim: claim("product-routing-defect", 1),
|
|
4319
|
+
classification: "product",
|
|
4320
|
+
triage: [
|
|
4321
|
+
{
|
|
4322
|
+
layer: "product",
|
|
4323
|
+
status: "unresolved",
|
|
4324
|
+
reason: "The product selected an unrelated Skill for an ordinary Work turn.",
|
|
4325
|
+
evidenceRefs: [ref("product-routing")]
|
|
4326
|
+
}
|
|
4327
|
+
],
|
|
4328
|
+
reason: "Product routing must be corrected before behavioral Evaluation.",
|
|
4329
|
+
nextAuthority: "product_team"
|
|
4330
|
+
});
|
|
4331
|
+
var taskset = createHarnessEvaluationReviewReceipt({
|
|
4332
|
+
...withoutHash2(noAction),
|
|
4333
|
+
id: "review-taskset",
|
|
4334
|
+
selectedEvidence: [
|
|
4335
|
+
evidence("failure-1", "work_outcome"),
|
|
4336
|
+
evidence("failure-2", "work_outcome"),
|
|
4337
|
+
evidence("failure-3", "work_outcome")
|
|
4338
|
+
],
|
|
4339
|
+
claim: claim("search-budget-allocation"),
|
|
4340
|
+
classification: "taskset",
|
|
4341
|
+
triage: [
|
|
4342
|
+
{
|
|
4343
|
+
layer: "harness",
|
|
4344
|
+
status: "unresolved",
|
|
4345
|
+
reason: "The active research Skill did not resolve three independent failures.",
|
|
4346
|
+
evidenceRefs: [ref("failure-1"), ref("failure-2"), ref("failure-3")]
|
|
4347
|
+
}
|
|
4348
|
+
],
|
|
4349
|
+
reason: "The repeated behavioral claim now needs controlled measurement.",
|
|
4350
|
+
nextAuthority: "human_review",
|
|
4351
|
+
tasksetProposal: ref("taskset-proposal"),
|
|
4352
|
+
maxEstimatedCostUsd: 2
|
|
4353
|
+
});
|
|
4354
|
+
var blockedRl = createModelImprovementQualificationReceipt({
|
|
4355
|
+
schemaVersion: "openpond.modelImprovementQualificationReceipt.v1",
|
|
4356
|
+
id: "qualification-rl-blocked",
|
|
4357
|
+
review: reference(taskset),
|
|
4358
|
+
harnessRelease,
|
|
4359
|
+
tasksetRelease: ref("taskset-release"),
|
|
4360
|
+
baselineEvaluation: ref("baseline-evaluation"),
|
|
4361
|
+
model: modelRef(),
|
|
4362
|
+
environmentHash: contentHash("environment"),
|
|
4363
|
+
toolContractHash: contentHash("tools"),
|
|
4364
|
+
permissionContractHash: contentHash("permissions"),
|
|
4365
|
+
policyHash: contentHash("policy"),
|
|
4366
|
+
verifierRef: ref("verifier"),
|
|
4367
|
+
sourcePolicies: [sourcePolicy],
|
|
4368
|
+
trainingEvidenceRefs: [ref("training-evidence")],
|
|
4369
|
+
frozenEvaluationEvidenceRefs: [ref("frozen-evidence")],
|
|
4370
|
+
privacyApproval: ref("privacy-approval"),
|
|
4371
|
+
budgetApproval: ref("budget-approval"),
|
|
4372
|
+
maximumCostUsd: 25,
|
|
4373
|
+
signal: {
|
|
4374
|
+
kind: "scalar_reward",
|
|
4375
|
+
strength: "weak",
|
|
4376
|
+
calibrated: true,
|
|
4377
|
+
confounded: false,
|
|
4378
|
+
variance: 0,
|
|
4379
|
+
evidenceRefs: [ref("reward-audit")]
|
|
4380
|
+
},
|
|
4381
|
+
decision: "no_training",
|
|
4382
|
+
reasons: ["The observed reward is constant and cannot support RL."],
|
|
4383
|
+
createdAt,
|
|
4384
|
+
metadata: { blockedMethod: "rl" }
|
|
4385
|
+
});
|
|
4386
|
+
var qualifiedRl = createModelImprovementQualificationReceipt({
|
|
4387
|
+
...withoutHash2(blockedRl),
|
|
4388
|
+
id: "qualification-rl-qualified",
|
|
4389
|
+
signal: {
|
|
4390
|
+
kind: "scalar_reward",
|
|
4391
|
+
strength: "usable",
|
|
4392
|
+
calibrated: true,
|
|
4393
|
+
confounded: false,
|
|
4394
|
+
variance: 0.18,
|
|
4395
|
+
evidenceRefs: [ref("reward-audit")]
|
|
4396
|
+
},
|
|
4397
|
+
decision: "rl",
|
|
4398
|
+
reasons: [
|
|
4399
|
+
"The frozen baseline has usable variance and a calibrated sequential reward."
|
|
4400
|
+
]
|
|
4401
|
+
});
|
|
4402
|
+
var modelImprovement = createHarnessEvaluationReviewReceipt({
|
|
4403
|
+
...withoutHash2(noAction),
|
|
4404
|
+
id: "review-model-improvement",
|
|
4405
|
+
selectedEvidence: [
|
|
4406
|
+
evidence("baseline-evaluation", "evaluation"),
|
|
4407
|
+
evidence("qualification-rl-qualified", "training_qualification")
|
|
4408
|
+
],
|
|
4409
|
+
claim: claim("search-budget-allocation"),
|
|
4410
|
+
classification: "model_improvement",
|
|
4411
|
+
triage: [
|
|
4412
|
+
{
|
|
4413
|
+
layer: "model",
|
|
4414
|
+
status: "unresolved",
|
|
4415
|
+
reason: "Harness, runtime, product, retrieval, and tool triage left a qualified sequential policy gap.",
|
|
4416
|
+
evidenceRefs: [ref("baseline-evaluation"), reference(qualifiedRl)]
|
|
4417
|
+
}
|
|
4418
|
+
],
|
|
4419
|
+
reason: "The qualified claim may proceed to a separately approved managed-training plan.",
|
|
4420
|
+
nextAuthority: "training_system",
|
|
4421
|
+
tasksetProposal: ref("taskset-proposal"),
|
|
4422
|
+
evaluation: ref("baseline-evaluation"),
|
|
4423
|
+
trainingQualification: reference(qualifiedRl),
|
|
4424
|
+
maxEstimatedCostUsd: 25
|
|
4425
|
+
});
|
|
4426
|
+
function ref(id) {
|
|
4427
|
+
return { id, contentHash: contentHash(id) };
|
|
4428
|
+
}
|
|
4429
|
+
function reference(value) {
|
|
4430
|
+
return { id: value.id, contentHash: value.contentHash };
|
|
4431
|
+
}
|
|
4432
|
+
function withoutHash2(value) {
|
|
4433
|
+
const { contentHash: _contentHash, ...content } = value;
|
|
4434
|
+
return content;
|
|
4435
|
+
}
|
|
4436
|
+
function modelRef() {
|
|
4437
|
+
return {
|
|
4438
|
+
provider: "openpond",
|
|
4439
|
+
model: "openpond-chat",
|
|
4440
|
+
revision: "model-revision-1",
|
|
4441
|
+
artifactHash: contentHash("model-artifact"),
|
|
4442
|
+
tokenizerRevision: "tokenizer-1",
|
|
4443
|
+
chatTemplateHash: contentHash("chat-template")
|
|
4444
|
+
};
|
|
4445
|
+
}
|
|
4446
|
+
|
|
3962
4447
|
export {
|
|
3963
4448
|
CONNECTED_APP_PROVIDER_TOOL_NAMES,
|
|
3964
4449
|
CONNECTED_APP_TOOL_CALL_ENDPOINT,
|
|
@@ -4000,6 +4485,10 @@ export {
|
|
|
4000
4485
|
createHarnessImprovementProposal,
|
|
4001
4486
|
createHarnessTargetedValidationReceipt,
|
|
4002
4487
|
createHarnessAdvanceReceipt,
|
|
4488
|
+
HarnessReviewSourcePolicyRefSchema,
|
|
4489
|
+
HarnessReviewWatermarkSchema,
|
|
4490
|
+
HarnessEvaluationReviewReceiptSchema,
|
|
4491
|
+
createHarnessEvaluationReviewReceipt,
|
|
4003
4492
|
ToolDeclarationSchema,
|
|
4004
4493
|
CapabilityRequirementSchema,
|
|
4005
4494
|
AgentSnapshotSchema,
|
|
@@ -4032,6 +4521,7 @@ export {
|
|
|
4032
4521
|
sameWorkspaceRevision,
|
|
4033
4522
|
stableId2 as stableId,
|
|
4034
4523
|
AttemptReceiptSchema,
|
|
4524
|
+
EvaluationResultSchema,
|
|
4035
4525
|
createRunManifest,
|
|
4036
4526
|
createAttemptReceipt,
|
|
4037
4527
|
aggregateEvaluationReceipts,
|
|
@@ -4050,5 +4540,7 @@ export {
|
|
|
4050
4540
|
createWorkFeedbackReceipt,
|
|
4051
4541
|
workEvidenceReceiptRef,
|
|
4052
4542
|
WorkEvidencePolicyStateSchema,
|
|
4053
|
-
classifyWorkEvidence
|
|
4543
|
+
classifyWorkEvidence,
|
|
4544
|
+
ModelImprovementQualificationReceiptSchema,
|
|
4545
|
+
createModelImprovementQualificationReceipt
|
|
4054
4546
|
};
|