openpond 0.0.51 → 0.0.52

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (95) hide show
  1. package/dist/chunks/{app-layer-DCUTQ4TM.js → app-layer-T26MAAGI.js} +2 -2
  2. package/dist/chunks/{app-server-runtime-Q3IGU5YB.js → app-server-runtime-2P2727X6.js} +6 -6
  3. package/dist/chunks/{apps-JDRCDAPE.js → apps-7KLTYKHN.js} +2 -2
  4. package/dist/chunks/{chunk-LYEXNSCO.js → chunk-3UPQUD6Y.js} +2 -2
  5. package/dist/chunks/{chunk-3PYU3XCS.js → chunk-55KT7K5P.js} +1 -2
  6. package/dist/chunks/{chunk-PFNX5XOT.js → chunk-6QWTEC2D.js} +1 -1
  7. package/dist/chunks/{chunk-J4LLKY6E.js → chunk-C4DAHPVI.js} +1 -1
  8. package/dist/chunks/{chunk-IGSEFSMA.js → chunk-EIRW6Z6S.js} +1 -1
  9. package/dist/chunks/{chunk-V3HADQL5.js → chunk-FFD2YSZR.js} +1 -1
  10. package/dist/chunks/{chunk-R4PRXXTC.js → chunk-JSRCXDIL.js} +32 -32
  11. package/dist/chunks/{chunk-UDTHREHB.js → chunk-KNKZSVGO.js} +81 -8
  12. package/dist/chunks/{chunk-CSN3DLWJ.js → chunk-M2OSF2YN.js} +1 -1
  13. package/dist/chunks/{chunk-35OYRVVH.js → chunk-US7DOPAH.js} +19 -4
  14. package/dist/chunks/{cli-2FTM4SXH.js → cli-Z2CULBCU.js} +5 -5
  15. package/dist/chunks/{core-commands-5BRR7TU4.js → core-commands-SND555SC.js} +3 -3
  16. package/dist/chunks/{desktop-test-EQ2BGZOM.js → desktop-test-WSWXNSFS.js} +2 -2
  17. package/dist/chunks/{extension-V3NZ7MU7.js → extension-3IKF6YQE.js} +1 -1
  18. package/dist/chunks/{help-WJBMX37U.js → help-DHGNXUA4.js} +1 -1
  19. package/dist/chunks/{opchat-W7DSKNNB.js → opchat-XYKT5KKO.js} +2 -2
  20. package/dist/chunks/{organizations-RHOF5H4Z.js → organizations-HKVC4CSA.js} +4 -4
  21. package/dist/chunks/{profile-B62ZWPTQ.js → profile-PLWXQHKF.js} +2 -2
  22. package/dist/chunks/{project-agent-QA25HHDV.js → project-agent-JNSP56KG.js} +2 -2
  23. package/dist/chunks/{sandbox-command-JFEVPU4X.js → sandbox-command-Z6LNNSSL.js} +2 -2
  24. package/dist/chunks/{sandbox-template-GEWTAWAR.js → sandbox-template-FVW2UTI4.js} +3 -3
  25. package/dist/chunks/{src-P2BL3KCV.js → src-2TP42H76.js} +3 -3
  26. package/dist/chunks/{src-JOTV67WO.js → src-R5VKB33T.js} +387 -599
  27. package/dist/chunks/{teams-bot-ODSMRNE4.js → teams-bot-FNPT4Q2B.js} +2 -2
  28. package/dist/chunks/{workspaces-7YHBJG5P.js → workspaces-YLNB6DS7.js} +2 -2
  29. package/dist/cli.js +13 -13
  30. package/dist/web/assets/{AppDialog-DylZeiI-.js → AppDialog-BEKS_Ixr.js} +1 -1
  31. package/dist/web/assets/{AppsView-Bz6C0a1u.js → AppsView-CbuT2sT2.js} +1 -1
  32. package/dist/web/assets/{BrowserSidebar-xXFxZP2-.js → BrowserSidebar-DgqVeW5u.js} +1 -1
  33. package/dist/web/assets/{CommandMenu-C9k5VdaR.js → CommandMenu-CQcaEB76.js} +1 -1
  34. package/dist/web/assets/{CommunityView-DcJlUNgd.js → CommunityView-CYX6iFcg.js} +1 -1
  35. package/dist/web/assets/{ComposerCreateImproveStrip-Yt3irJVo.js → ComposerCreateImproveStrip-DY4SQbL4.js} +1 -1
  36. package/dist/web/assets/{GetStartedView-BhD0Ij7c.js → GetStartedView-DnxeZaBB.js} +1 -1
  37. package/dist/web/assets/{LabModelVersionDetailPage-B-OoSJYT.js → LabModelVersionDetailPage-DYTfYGWv.js} +1 -1
  38. package/dist/web/assets/{LabSkillSidebar-dqdSTmHG.js → LabSkillSidebar-BpD-kjP_.js} +1 -1
  39. package/dist/web/assets/{LabsRoute-BMPI4vB8.js → LabsRoute-DVC7eZCs.js} +3 -3
  40. package/dist/web/assets/{MainChatThread-B0ajKiNL.js → MainChatThread-DTiX_8MC.js} +2 -2
  41. package/dist/web/assets/{MainPane-COs4xUGo.js → MainPane-CNoHJkXy.js} +3 -3
  42. package/dist/web/assets/{MarkdownText-CzZ60aWs.js → MarkdownText-_JhfBcQL.js} +1 -1
  43. package/dist/web/assets/{Messages-DGrK4iIj.js → Messages-Bo0-Uvcz.js} +1 -1
  44. package/dist/web/assets/{NativeSkillSidebar-BkpgGLSO.js → NativeSkillSidebar-CHQ87FW2.js} +1 -1
  45. package/dist/web/assets/{NewProjectDialog-BOdCdNcG.js → NewProjectDialog-B-1KuU5u.js} +1 -1
  46. package/dist/web/assets/{OutputsPage-CSBWIFdi.js → OutputsPage-BSimpEpA.js} +1 -1
  47. package/dist/web/assets/{RightChatPanelStack-PZR6NYJj.js → RightChatPanelStack-B8ZOyuFO.js} +1 -1
  48. package/dist/web/assets/{ScheduledWorkPage-WXLnwju_.js → ScheduledWorkPage-Blwy1G8r.js} +1 -1
  49. package/dist/web/assets/{SettingsView-CQJoEx-y.js → SettingsView-CTLTakyn.js} +3 -3
  50. package/dist/web/assets/{TeamChatView-BRnuD-lV.js → TeamChatView-dN3fUCbZ.js} +1 -1
  51. package/dist/web/assets/{TerminalOverlay-B5m8j4_C.js → TerminalOverlay-CPVGeO7U.js} +1 -1
  52. package/dist/web/assets/{TrainingCreationPanel-CA1rt7XZ.js → TrainingCreationPanel-ettNNJBP.js} +1 -1
  53. package/dist/web/assets/{TrainingDraftPanel-C0p3GP1n.js → TrainingDraftPanel-DUJBuII1.js} +1 -1
  54. package/dist/web/assets/{UsageSettingsSection-CpIKj4p-.js → UsageSettingsSection-BOZR3JE_.js} +1 -1
  55. package/dist/web/assets/{WorkspaceDiffPanel-BXk8mdZB.js → WorkspaceDiffPanel-99MjzMkW.js} +3 -3
  56. package/dist/web/assets/{WorkspaceEnvironmentMenu-BO60uNVj.js → WorkspaceEnvironmentMenu-C3s3mTyu.js} +1 -1
  57. package/dist/web/assets/{WorkspaceGitDialogs-hWBfhgaP.js → WorkspaceGitDialogs-BvurXjNP.js} +1 -1
  58. package/dist/web/assets/{WorkspaceMonacoEditor-ULPyacH8.js → WorkspaceMonacoEditor-BCXg9mBQ.js} +3 -3
  59. package/dist/web/assets/{arrow-up-right-CkfCgokz.js → arrow-up-right-DRIT5z3q.js} +1 -1
  60. package/dist/web/assets/{chevron-up-B5URpgIH.js → chevron-up-Hglu-Zx3.js} +1 -1
  61. package/dist/web/assets/{circle-alert-CAvjRdlQ.js → circle-alert-Crh3Zr2m.js} +1 -1
  62. package/dist/web/assets/{cloud-upload-DeQEwz1o.js → cloud-upload-mSm9KIL4.js} +1 -1
  63. package/dist/web/assets/{cssMode-B4V5a3OX.js → cssMode-CkQ4x6JA.js} +1 -1
  64. package/dist/web/assets/{folder-DgS_U9Lq.js → folder-DMeODEQJ.js} +1 -1
  65. package/dist/web/assets/{folder-git-2-BTtbxKWf.js → folder-git-2-B8TV5i0V.js} +1 -1
  66. package/dist/web/assets/{folder-open-CqjIkaDK.js → folder-open-DXaun_C3.js} +1 -1
  67. package/dist/web/assets/{folder-plus-CsS0KPW5.js → folder-plus-CUbTwu82.js} +1 -1
  68. package/dist/web/assets/{git-branch-BN2ijBYt.js → git-branch-CUlgF0kK.js} +1 -1
  69. package/dist/web/assets/{git-commit-horizontal-CVsl4u1O.js → git-commit-horizontal-DAQfyhbS.js} +1 -1
  70. package/dist/web/assets/{htmlMode-DRksGWJA.js → htmlMode-BfyAplZo.js} +1 -1
  71. package/dist/web/assets/index-Blindd8c.js +1 -0
  72. package/dist/web/assets/{index-Cxq5q-6B.js → index-CDJX5ZxL.js} +3 -3
  73. package/dist/web/assets/{info-DwVXRQli.js → info-Ll3PkeGE.js} +1 -1
  74. package/dist/web/assets/{jsonMode-eJvgtAy-.js → jsonMode-C9Vibh0P.js} +1 -1
  75. package/dist/web/assets/{lspLanguageFeatures-DCk7zUI-.js → lspLanguageFeatures-CFoA9--Z.js} +1 -1
  76. package/dist/web/assets/{monaco.contribution-B02C5mDa.js → monaco.contribution-C2q-tDuk.js} +2 -2
  77. package/dist/web/assets/{monaco.contribution-DWVguH46.js → monaco.contribution-C8tRjUqy.js} +2 -2
  78. package/dist/web/assets/{monaco.contribution-WGT1eYvk.js → monaco.contribution-Co4ZPnCC.js} +2 -2
  79. package/dist/web/assets/{monaco.contribution-CPnN-lZN.js → monaco.contribution-vWDjhuCI.js} +2 -2
  80. package/dist/web/assets/{play-DGCzapdE.js → play-lC5KpmRD.js} +1 -1
  81. package/dist/web/assets/{python-DRPzruv8.js → python-I5xkyvMs.js} +1 -1
  82. package/dist/web/assets/{refresh-cw-BKF4NhDI.js → refresh-cw-8a_wz4tX.js} +1 -1
  83. package/dist/web/assets/{save-DQXtfx_4.js → save-Dj9ex3Ce.js} +1 -1
  84. package/dist/web/assets/{square-Dp5eGICS.js → square-DfEM9XgY.js} +1 -1
  85. package/dist/web/assets/{square-pen-j-mZ_UkO.js → square-pen-CgX3HoJ6.js} +1 -1
  86. package/dist/web/assets/{toggleHighContrast-wDmA5YPb.js → toggleHighContrast-BkNzz1Bg.js} +1 -1
  87. package/dist/web/assets/{tsMode-DELTv39m.js → tsMode-CNwaV0ZJ.js} +1 -1
  88. package/dist/web/assets/{upload-_rNli8wK.js → upload-Co5zYY9Z.js} +1 -1
  89. package/dist/web/assets/{useLocalAgentSchedules-DRVUfTbv.js → useLocalAgentSchedules-QY4_XS2F.js} +1 -1
  90. package/dist/web/assets/{wifi-off-uKPwd8A7.js → wifi-off-CFzyn7Tf.js} +1 -1
  91. package/dist/web/assets/{workers-CXbluSB3.js → workers-DqFxl1zu.js} +1 -1
  92. package/dist/web/assets/{yaml-BMqi-3ui.js → yaml-Bbm66uDK.js} +1 -1
  93. package/dist/web/index.html +1 -1
  94. package/package.json +1 -1
  95. package/dist/web/assets/index-SKvMOpII.js +0 -1
@@ -117,10 +117,10 @@ import {
117
117
  workspaceHasHead,
118
118
  workspaceToolExperienceBlocker,
119
119
  writeWorkspaceFile
120
- } from "./chunk-UDTHREHB.js";
120
+ } from "./chunk-KNKZSVGO.js";
121
121
  import {
122
122
  runOpenPondServerCli
123
- } from "./chunk-LYEXNSCO.js";
123
+ } from "./chunk-3UPQUD6Y.js";
124
124
  import {
125
125
  createAppServer,
126
126
  runAgentCompaction
@@ -138,7 +138,7 @@ import {
138
138
  streamOpenPondHostedChatTurn,
139
139
  switchOpenPondAccount,
140
140
  updateOpenPondAccountConfig
141
- } from "./chunk-IGSEFSMA.js";
141
+ } from "./chunk-EIRW6Z6S.js";
142
142
  import {
143
143
  appWorkspacePaths,
144
144
  checkWorkspaceGitAvailability,
@@ -166,7 +166,7 @@ import {
166
166
  runCommand,
167
167
  startMacOSCommandLineToolsInstall,
168
168
  workspaceImageContentType
169
- } from "./chunk-V3HADQL5.js";
169
+ } from "./chunk-FFD2YSZR.js";
170
170
  import {
171
171
  APP_PREFERENCES_CACHE_KEY,
172
172
  APP_PREFERENCES_CACHE_TYPE,
@@ -178,7 +178,7 @@ import {
178
178
  isCliEntrypoint,
179
179
  now,
180
180
  textFromUnknown
181
- } from "./chunk-J4LLKY6E.js";
181
+ } from "./chunk-C4DAHPVI.js";
182
182
  import {
183
183
  AccountStateSchema,
184
184
  AdapterValidationReceiptSchema,
@@ -244,7 +244,6 @@ import {
244
244
  ModelBindingRoleSchema,
245
245
  ModelBindingSchema,
246
246
  ModelEvaluationReceiptSchema,
247
- ModelEvaluationStopReceiptSchema,
248
247
  ModelProjectSchema,
249
248
  ModelRunDraftSchema,
250
249
  ModelRunSchema,
@@ -412,7 +411,7 @@ import {
412
411
  workFormatCapabilityForContentType,
413
412
  workspacePathFromLocalPathWorkspaceId,
414
413
  writeSourceUploadCache
415
- } from "./chunk-3PYU3XCS.js";
414
+ } from "./chunk-55KT7K5P.js";
416
415
  import {
417
416
  AttemptReceiptContentSchema,
418
417
  AttemptReceiptSchema,
@@ -452,7 +451,7 @@ import {
452
451
  workEvidenceReceiptRef,
453
452
  workSourceOpaqueRef,
454
453
  workWorkspaceOpaqueRef
455
- } from "./chunk-35OYRVVH.js";
454
+ } from "./chunk-US7DOPAH.js";
456
455
  import {
457
456
  external_exports,
458
457
  yaml
@@ -2706,7 +2705,7 @@ var require_websocket = __commonJS({
2706
2705
  var http = __require("http");
2707
2706
  var net = __require("net");
2708
2707
  var tls = __require("tls");
2709
- var { randomBytes: randomBytes2, createHash: createHash18 } = __require("crypto");
2708
+ var { randomBytes: randomBytes2, createHash: createHash17 } = __require("crypto");
2710
2709
  var { Duplex, Readable } = __require("stream");
2711
2710
  var { URL: URL2 } = __require("url");
2712
2711
  var PerMessageDeflate2 = require_permessage_deflate();
@@ -3374,7 +3373,7 @@ var require_websocket = __commonJS({
3374
3373
  abortHandshake(websocket, socket, "Invalid Upgrade header");
3375
3374
  return;
3376
3375
  }
3377
- const digest = createHash18("sha1").update(key + GUID).digest("base64");
3376
+ const digest = createHash17("sha1").update(key + GUID).digest("base64");
3378
3377
  if (res.headers["sec-websocket-accept"] !== digest) {
3379
3378
  abortHandshake(websocket, socket, "Invalid Sec-WebSocket-Accept header");
3380
3379
  return;
@@ -3743,7 +3742,7 @@ var require_websocket_server = __commonJS({
3743
3742
  var EventEmitter = __require("events");
3744
3743
  var http = __require("http");
3745
3744
  var { Duplex } = __require("stream");
3746
- var { createHash: createHash18 } = __require("crypto");
3745
+ var { createHash: createHash17 } = __require("crypto");
3747
3746
  var extension2 = require_extension();
3748
3747
  var PerMessageDeflate2 = require_permessage_deflate();
3749
3748
  var subprotocol2 = require_subprotocol();
@@ -4050,7 +4049,7 @@ var require_websocket_server = __commonJS({
4050
4049
  );
4051
4050
  }
4052
4051
  if (this._state > RUNNING) return abortHandshake(socket, 503);
4053
- const digest = createHash18("sha1").update(key + GUID).digest("base64");
4052
+ const digest = createHash17("sha1").update(key + GUID).digest("base64");
4054
4053
  const headers = [
4055
4054
  "HTTP/1.1 101 Switching Protocols",
4056
4055
  "Upgrade: websocket",
@@ -48851,6 +48850,11 @@ function errorMessage4(error) {
48851
48850
 
48852
48851
  // ../server/src/training/harness-refiner-benchmark-protocol.ts
48853
48852
  function createHarnessRefinerExecutionPlan(input) {
48853
+ if (input.seeds.length !== 1 || input.repetitions !== 1) {
48854
+ throw new Error(
48855
+ "Sequential Harness Refiner benchmarks require one admitted seed and one trajectory repetition."
48856
+ );
48857
+ }
48854
48858
  const benchmark = input.taskset.benchmark;
48855
48859
  if (!benchmark) throw new Error("Harness Refiner Taskset has no benchmark definition.");
48856
48860
  const heldOut = taskIdsForSplit(input.taskset, benchmark.evaluationSplit);
@@ -49029,7 +49033,7 @@ import { promises as fs23 } from "node:fs";
49029
49033
  import path56 from "node:path";
49030
49034
  function createResultManifest(input) {
49031
49035
  const core = {
49032
- schemaVersion: "openpond.harnessRefinerBenchmarkResult.v1",
49036
+ schemaVersion: "openpond.harnessRefinerBenchmarkResult.v2",
49033
49037
  id: `benchmark-result-${input.modelRunId}`,
49034
49038
  modelRunId: input.modelRunId,
49035
49039
  benchmarkId: "harness-refiner",
@@ -49081,7 +49085,7 @@ async function loadLatestManagedResult(storeDir, modelRunId) {
49081
49085
  const value = JSON.parse(
49082
49086
  await fs23.readFile(path56.join(root, entry), "utf8")
49083
49087
  );
49084
- return value.schemaVersion === "openpond.harnessRefinerBenchmarkResult.v1" && value.modelRunId === modelRunId ? value : null;
49088
+ return value.schemaVersion === "openpond.harnessRefinerBenchmarkResult.v2" && value.modelRunId === modelRunId ? value : null;
49085
49089
  }));
49086
49090
  return manifests.filter((value) => value !== null).sort((left, right) => right.createdAt.localeCompare(left.createdAt))[0] ?? null;
49087
49091
  }
@@ -49180,7 +49184,6 @@ async function ensureBaseVersion(input) {
49180
49184
  }
49181
49185
 
49182
49186
  // ../server/src/training/harness-refiner-benchmark-service-support.ts
49183
- import { createHash as createHash15 } from "node:crypto";
49184
49187
  import { promises as fs24 } from "node:fs";
49185
49188
  import path57 from "node:path";
49186
49189
  function completedStage(input) {
@@ -49322,6 +49325,30 @@ async function loadCompletedBenchmarkStage(input) {
49322
49325
  }));
49323
49326
  return { run, attempts: evidence };
49324
49327
  }
49328
+ async function loadBenchmarkAttemptEvidenceByIds(input) {
49329
+ const wanted = new Set(input.attemptIds);
49330
+ const [attempts, grades] = await Promise.all([
49331
+ input.store.listTaskAttempts(input.tasksetId),
49332
+ input.store.listGradeResultsForTaskset(input.tasksetId)
49333
+ ]);
49334
+ const selected = attempts.filter((attempt) => wanted.has(attempt.id));
49335
+ if (selected.length !== wanted.size) {
49336
+ throw new Error(
49337
+ `Sequential adaptation evidence has ${selected.length}/${wanted.size} attempts.`
49338
+ );
49339
+ }
49340
+ const gradesByAttempt = new Map(grades.map((grade) => [grade.attemptId, grade]));
49341
+ return Promise.all(selected.map(async (attempt) => {
49342
+ const grade = gradesByAttempt.get(attempt.id);
49343
+ if (!grade) throw new Error(`Attempt ${attempt.id} has no durable grade.`);
49344
+ return {
49345
+ attempt,
49346
+ grade,
49347
+ artifacts: await input.store.listTaskAttemptArtifacts({ attemptId: attempt.id }),
49348
+ receiptContentHash: portableReceiptContentHash(attempt)
49349
+ };
49350
+ }));
49351
+ }
49325
49352
  function attemptHarnessReleaseHash(attempt) {
49326
49353
  const capability = objectRecord(attempt.metadata.harnessCapabilityReceipt);
49327
49354
  const release = objectRecord(capability?.harnessRelease);
@@ -49605,47 +49632,6 @@ function attemptUsageSummary(rawUsage) {
49605
49632
  totalTokens: usage.totalTokens
49606
49633
  };
49607
49634
  }
49608
- function attemptToolFailureCount(result) {
49609
- const value = result.attempt.output.toolFailureCount;
49610
- return typeof value === "number" && Number.isFinite(value) && value >= 0 ? Math.trunc(value) : 0;
49611
- }
49612
- async function benchmarkToolFailureEvidence(input) {
49613
- if (input.expectedCount === 0) return { failures: [], omittedCount: 0 };
49614
- const trace = input.artifacts.find((artifact) => artifact.kind === "runtime_trace");
49615
- if (!trace) {
49616
- throw new Error(`Attempt ${input.attemptId} has tool failures but no runtime trace.`);
49617
- }
49618
- const bytes2 = await fs24.readFile(trace.path);
49619
- const digest = createHash15("sha256").update(bytes2).digest("hex");
49620
- if (digest !== trace.sha256) {
49621
- throw new Error(`Attempt ${input.attemptId} runtime trace failed hash validation.`);
49622
- }
49623
- const parsed = objectRecord(JSON.parse(bytes2.toString("utf8")));
49624
- const steps = Array.isArray(parsed?.steps) ? parsed.steps.map((step) => objectRecord(step)).filter(Boolean) : [];
49625
- const failures = steps.flatMap((step, index) => {
49626
- if (step?.kind !== "tool" || step.ok !== false) return [];
49627
- const toolName = typeof step.name === "string" && step.name.trim() ? step.name.trim() : "unknown_tool";
49628
- const recoveredLater = steps.slice(index + 1).some(
49629
- (candidate2) => candidate2?.kind === "tool" && candidate2.name === toolName && candidate2.ok === true
49630
- );
49631
- return [{
49632
- toolName,
49633
- turn: typeof step.turn === "number" && Number.isInteger(step.turn) ? step.turn : null,
49634
- detail: typeof step.output === "string" ? step.output.slice(0, 1e3) : "Tool failed without textual output.",
49635
- recoveredLater
49636
- }];
49637
- });
49638
- if (failures.length !== input.expectedCount) {
49639
- throw new Error(
49640
- `Attempt ${input.attemptId} tool-failure count drifted between its receipt and runtime trace.`
49641
- );
49642
- }
49643
- const visible = failures.slice(0, 20);
49644
- return {
49645
- failures: visible,
49646
- omittedCount: failures.length - visible.length
49647
- };
49648
- }
49649
49635
  function emptyUsageCategory() {
49650
49636
  return { inputTokens: 0, outputTokens: 0, totalTokens: 0, costUsd: null };
49651
49637
  }
@@ -49760,7 +49746,7 @@ function comparisonInvalidReasons(input) {
49760
49746
  for (const [label, run] of [
49761
49747
  ["held-out baseline", input.baseline],
49762
49748
  ["adaptation baseline", input.adaptation],
49763
- ["candidate adaptation replay", input.candidateAdaptation],
49749
+ ["sequential adaptation treatment", input.candidateAdaptation],
49764
49750
  ["held-out candidate", input.candidate]
49765
49751
  ]) {
49766
49752
  if (run.terminalCount !== run.attemptCount) {
@@ -49773,7 +49759,7 @@ function comparisonInvalidReasons(input) {
49773
49759
  if (!input.harnessChanged) reasons.push("The candidate Harness is unchanged.");
49774
49760
  if (!input.lineageValid) reasons.push("Candidate Harness lineage is incomplete.");
49775
49761
  if (input.candidateAdaptation.passedCount !== input.candidateAdaptation.attemptCount) {
49776
- reasons.push("Candidate adaptation replay did not pass every case.");
49762
+ reasons.push("Sequential adaptation treatment did not pass every case.");
49777
49763
  }
49778
49764
  if (input.candidate.passedCount !== input.candidate.attemptCount) {
49779
49765
  reasons.push("Candidate held-out quality did not pass every case.");
@@ -49808,7 +49794,7 @@ async function resumeHarnessRefinerComparison(input) {
49808
49794
  budget,
49809
49795
  totalAttempts
49810
49796
  } = input;
49811
- const [baseline, adaptation, priorCandidateAdaptation, priorCandidate] = await Promise.all([
49797
+ const [baseline, adaptation, priorCandidate] = await Promise.all([
49812
49798
  loadCompletedBenchmarkStage({
49813
49799
  store: deps.store,
49814
49800
  modelRunId: modelRun.id,
@@ -49821,12 +49807,6 @@ async function resumeHarnessRefinerComparison(input) {
49821
49807
  tasksetId: taskset.id,
49822
49808
  plan: input.adaptationPlan
49823
49809
  }),
49824
- loadCompletedBenchmarkStage({
49825
- store: deps.store,
49826
- modelRunId: modelRun.id,
49827
- tasksetId: taskset.id,
49828
- plan: input.candidateAdaptationPlan
49829
- }),
49830
49810
  loadCompletedBenchmarkStage({
49831
49811
  store: deps.store,
49832
49812
  modelRunId: modelRun.id,
@@ -49838,6 +49818,12 @@ async function resumeHarnessRefinerComparison(input) {
49838
49818
  if (!priorManifest) {
49839
49819
  throw new Error("Comparison recovery requires the durable benchmark result manifest.");
49840
49820
  }
49821
+ const sequentialAttemptIds = modelRun.evaluationProgress?.accounting?.attempts.filter((attempt) => attempt.phase === "candidate_adaptation").map((attempt) => attempt.attemptId) ?? [];
49822
+ const candidateAdaptationAttempts = await loadBenchmarkAttemptEvidenceByIds({
49823
+ store: deps.store,
49824
+ tasksetId: taskset.id,
49825
+ attemptIds: sequentialAttemptIds
49826
+ });
49841
49827
  if (priorManifest.tasksetRelease.contentHash !== taskset.benchmark?.releaseHash || priorManifest.harness.baseline.contentHash !== baseline.run.harnessRelease.contentHash) {
49842
49828
  throw new Error("Comparison recovery manifest drifted from the admitted benchmark.");
49843
49829
  }
@@ -49922,12 +49908,6 @@ async function resumeHarnessRefinerComparison(input) {
49922
49908
  createdAt: deps.now()
49923
49909
  });
49924
49910
  };
49925
- const candidateAdaptation = await retryStage(
49926
- priorCandidateAdaptation,
49927
- input.candidateAdaptationPlan,
49928
- "candidate_adaptation",
49929
- "adaptation"
49930
- );
49931
49911
  const candidate2 = await retryStage(
49932
49912
  priorCandidate,
49933
49913
  input.candidatePlan,
@@ -49967,7 +49947,7 @@ async function resumeHarnessRefinerComparison(input) {
49967
49947
  baseline: baseline.run,
49968
49948
  adaptation: adaptation.run,
49969
49949
  refiner: priorManifest.refiner,
49970
- candidateAdaptation: candidateAdaptation.run,
49950
+ candidateAdaptation: priorManifest.candidateAdaptation,
49971
49951
  candidate: candidate2.run,
49972
49952
  comparison,
49973
49953
  executionPlan,
@@ -49987,7 +49967,7 @@ async function resumeHarnessRefinerComparison(input) {
49987
49967
  const infrastructureValid = benchmarkAttemptsInfrastructureValid([
49988
49968
  ...baseline.attempts,
49989
49969
  ...adaptation.attempts,
49990
- ...candidateAdaptation.attempts,
49970
+ ...candidateAdaptationAttempts,
49991
49971
  ...candidate2.attempts
49992
49972
  ]);
49993
49973
  const harnessChanged = baseline.run.harnessRelease.contentHash !== candidate2.run.harnessRelease.contentHash;
@@ -49996,7 +49976,7 @@ async function resumeHarnessRefinerComparison(input) {
49996
49976
  baseline: baseline.run,
49997
49977
  adaptation: adaptation.run,
49998
49978
  candidate: candidate2.run,
49999
- candidateAdaptation: candidateAdaptation.run,
49979
+ candidateAdaptation: priorManifest.candidateAdaptation,
50000
49980
  harnessChanged,
50001
49981
  lineageValid: lineage.valid,
50002
49982
  infrastructureValid
@@ -50004,7 +49984,7 @@ async function resumeHarnessRefinerComparison(input) {
50004
49984
  const invalidReasons = comparisonInvalidReasons({
50005
49985
  baseline: baseline.run,
50006
49986
  adaptation: adaptation.run,
50007
- candidateAdaptation: candidateAdaptation.run,
49987
+ candidateAdaptation: priorManifest.candidateAdaptation,
50008
49988
  candidate: candidate2.run,
50009
49989
  harnessChanged,
50010
49990
  lineageValid: lineage.valid,
@@ -50030,8 +50010,8 @@ async function resumeHarnessRefinerComparison(input) {
50030
50010
  baseline: { id: baseline.run.id, contentHash: baseline.run.contentHash },
50031
50011
  adaptation: { id: adaptation.run.id, contentHash: adaptation.run.contentHash },
50032
50012
  candidateAdaptation: {
50033
- id: candidateAdaptation.run.id,
50034
- contentHash: candidateAdaptation.run.contentHash
50013
+ id: priorManifest.candidateAdaptation.id,
50014
+ contentHash: priorManifest.candidateAdaptation.contentHash
50035
50015
  },
50036
50016
  refiner: {
50037
50017
  id: priorManifest.refiner.id,
@@ -50045,10 +50025,10 @@ async function resumeHarnessRefinerComparison(input) {
50045
50025
  baselinePassRate: comparison.baselinePassRate,
50046
50026
  candidatePassRate: comparison.candidatePassRate,
50047
50027
  adaptationBaselinePassRate: adaptation.run.passedCount / adaptation.run.attemptCount,
50048
- adaptationCandidatePassRate: candidateAdaptation.run.passedCount / candidateAdaptation.run.attemptCount,
50049
- adaptationCandidatePassed: candidateAdaptation.run.passedCount === candidateAdaptation.run.attemptCount,
50028
+ adaptationCandidatePassRate: priorManifest.candidateAdaptation.passedCount / priorManifest.candidateAdaptation.attemptCount,
50029
+ adaptationCandidatePassed: priorManifest.candidateAdaptation.passedCount === priorManifest.candidateAdaptation.attemptCount,
50050
50030
  heldOutCandidatePassed: candidate2.run.passedCount === candidate2.run.attemptCount,
50051
- passed: comparison.qualityPassed && candidateAdaptation.run.passedCount === candidateAdaptation.run.attemptCount && candidate2.run.passedCount === candidate2.run.attemptCount && infrastructureValid && terminalClassification !== "infrastructure_failure"
50031
+ passed: comparison.qualityPassed && priorManifest.candidateAdaptation.passedCount === priorManifest.candidateAdaptation.attemptCount && candidate2.run.passedCount === candidate2.run.attemptCount && infrastructureValid && terminalClassification !== "infrastructure_failure"
50052
50032
  },
50053
50033
  foregroundTokenDelta: comparison.foregroundTokenDelta,
50054
50034
  foregroundTokenDeltaPercent: comparison.foregroundTokenDeltaPercent,
@@ -50090,385 +50070,307 @@ async function resumeHarnessRefinerComparison(input) {
50090
50070
  }));
50091
50071
  }
50092
50072
 
50093
- // ../server/src/training/harness-refiner-benchmark-cohort-evidence.ts
50094
- var MAX_REQUEST_CHARACTERS = 2500;
50095
- var MAX_OUTPUT_CHARACTERS = 4e3;
50096
- var BEHAVIOR_FAMILY_TAGS = [
50097
- "artifact-verification",
50098
- "research-efficiency",
50099
- "constraint-following"
50100
- ];
50101
- function boundedText(value, limit) {
50102
- const text3 = typeof value === "string" ? value : "";
50103
- return {
50104
- text: text3.slice(0, limit),
50105
- truncated: text3.length > limit
50106
- };
50107
- }
50108
- function taskBehaviorFamily(task) {
50109
- const families = BEHAVIOR_FAMILY_TAGS.filter((tag) => task.tags.includes(tag));
50110
- if (families.length !== 1) {
50111
- throw new Error(`Adaptation task ${task.id} has no behavior-family contract.`);
50112
- }
50113
- return families[0];
50114
- }
50115
- function artifactResults(output) {
50116
- const requiredOutputs2 = Array.isArray(output.requiredOutputs) ? output.requiredOutputs : [];
50117
- return requiredOutputs2.flatMap((value) => {
50118
- if (!value || typeof value !== "object" || Array.isArray(value)) return [];
50119
- const item = value;
50120
- if (typeof item.path !== "string" || typeof item.mediaType !== "string" || typeof item.passed !== "boolean") return [];
50121
- return [{
50122
- path: item.path,
50123
- mediaType: item.mediaType,
50124
- passed: item.passed,
50125
- validationKinds: Array.isArray(item.validationKinds) ? item.validationKinds.filter((kind) => typeof kind === "string") : []
50126
- }];
50127
- });
50128
- }
50129
- function repeatedToolFailureGroups(attempts) {
50130
- const groups = /* @__PURE__ */ new Map();
50131
- for (const attempt of attempts) {
50132
- for (const failure of attempt.toolFailures) {
50133
- const group = groups.get(failure.toolName) ?? {
50134
- occurrences: 0,
50135
- taskIds: /* @__PURE__ */ new Set()
50136
- };
50137
- group.occurrences += 1;
50138
- group.taskIds.add(attempt.taskId);
50139
- groups.set(failure.toolName, group);
50140
- }
50141
- }
50142
- return [...groups.entries()].map(([toolName, group]) => ({
50143
- toolName,
50144
- distinctTaskCount: group.taskIds.size,
50145
- occurrenceCount: group.occurrences,
50146
- taskIds: [...group.taskIds]
50147
- })).filter((group) => group.distinctTaskCount >= 2).sort(
50148
- (left, right) => right.distinctTaskCount - left.distinctTaskCount || right.occurrenceCount - left.occurrenceCount || left.toolName.localeCompare(right.toolName)
50149
- );
50150
- }
50151
- function selectPrimaryEvidenceAnchor(attempts, repeatedGroups) {
50152
- const repeatedTool = repeatedGroups[0]?.toolName;
50153
- if (repeatedTool) {
50154
- const ranked = attempts.map((attempt, index) => ({
50155
- attempt,
50156
- index,
50157
- matchingFailures: attempt.toolFailures.filter(
50158
- (failure) => failure.toolName === repeatedTool
50159
- ).length
50160
- })).filter((item) => item.matchingFailures > 0).sort(
50161
- (left, right) => right.matchingFailures - left.matchingFailures || left.attempt.toolFailureCount - right.attempt.toolFailureCount || left.index - right.index
50162
- );
50163
- const selected2 = ranked[0]?.attempt;
50164
- if (selected2) {
50165
- return {
50166
- attemptId: selected2.attemptId,
50167
- taskId: selected2.taskId,
50168
- reason: "repeated_cross_task_tool_failure"
50169
- };
50170
- }
50171
- }
50172
- const failed = attempts.find((attempt) => !attempt.passed);
50173
- if (failed) {
50174
- return {
50175
- attemptId: failed.attemptId,
50176
- taskId: failed.taskId,
50177
- reason: "failed_grade"
50178
- };
50179
- }
50180
- const selected = [...attempts].sort(
50181
- (left, right) => right.toolFailureCount - left.toolFailureCount || right.latencyMs - left.latencyMs
50182
- )[0];
50183
- if (!selected) throw new Error("Adaptation cohort produced no Refiner evidence.");
50184
- return {
50185
- attemptId: selected.attemptId,
50186
- taskId: selected.taskId,
50187
- reason: "highest_signal_attempt"
50188
- };
50189
- }
50190
- async function buildHarnessRefinerBenchmarkCohortEvidence(input) {
50191
- const attempts = await Promise.all(input.adaptationAttempts.map(async (result) => {
50192
- const task = input.taskset.tasks.find(
50193
- (candidate2) => candidate2.id === result.attempt.taskId
50194
- );
50195
- if (!task || task.split !== "validation") {
50196
- throw new Error(`Adaptation evidence task ${result.attempt.taskId} is unavailable.`);
50197
- }
50198
- const toolFailureCount = attemptToolFailureCount(result);
50199
- const toolFailureEvidence = await benchmarkToolFailureEvidence({
50200
- attemptId: result.attempt.id,
50201
- artifacts: result.artifacts,
50202
- expectedCount: toolFailureCount
50203
- });
50204
- const request = boundedText(taskPrompt2(task), MAX_REQUEST_CHARACTERS);
50205
- const assistantOutput = boundedText(
50206
- result.attempt.output.text,
50207
- MAX_OUTPUT_CHARACTERS
50208
- );
50209
- return {
50210
- attemptId: result.attempt.id,
50211
- taskId: result.attempt.taskId,
50212
- behaviorFamily: taskBehaviorFamily(task),
50213
- attemptReceiptHash: result.receiptContentHash,
50214
- gradeHash: contentHash(result.grade),
50215
- passed: result.grade.passed,
50216
- score: result.grade.score,
50217
- failureClass: result.grade.failureClass,
50218
- feedback: [...result.grade.feedback],
50219
- request: request.text,
50220
- requestTruncated: request.truncated,
50221
- assistantOutput: assistantOutput.text,
50222
- assistantOutputTruncated: assistantOutput.truncated,
50223
- evaluationCriteria: task.expectedOutput,
50224
- artifactResults: artifactResults(result.attempt.output),
50225
- outputsPassed: typeof result.attempt.output.outputsPassed === "boolean" ? result.attempt.output.outputsPassed : null,
50226
- toolFailureCount,
50227
- toolFailures: toolFailureEvidence.failures,
50228
- omittedToolFailureCount: toolFailureEvidence.omittedCount,
50229
- latencyMs: result.attempt.latencyMs,
50230
- usage: attemptUsageSummary(result.attempt.metadata.usage)
50231
- };
50232
- }));
50233
- const families = /* @__PURE__ */ new Map();
50234
- for (const attempt of attempts) {
50235
- const family = families.get(attempt.behaviorFamily) ?? [];
50236
- family.push(attempt);
50237
- families.set(attempt.behaviorFamily, family);
50238
- }
50239
- const behaviorFamilies = [...families.entries()].map(([behaviorFamily, familyAttempts]) => ({
50240
- behaviorFamily,
50241
- attemptCount: familyAttempts.length,
50242
- passedCount: familyAttempts.filter((attempt) => attempt.passed).length,
50243
- failedTaskIds: familyAttempts.filter((attempt) => !attempt.passed).map((attempt) => attempt.taskId),
50244
- taskIds: familyAttempts.map((attempt) => attempt.taskId),
50245
- toolFailureCount: familyAttempts.reduce(
50246
- (total, attempt) => total + attempt.toolFailureCount,
50247
- 0
50248
- ),
50249
- tasksWithToolFailures: familyAttempts.filter(
50250
- (attempt) => attempt.toolFailureCount > 0
50251
- ).length
50252
- })).sort((left, right) => left.behaviorFamily.localeCompare(right.behaviorFamily));
50253
- const crossTaskToolFailureGroups = repeatedToolFailureGroups(attempts);
50254
- return {
50255
- schemaVersion: "openpond.harnessRefinerBenchmarkCohortEvidence.v2",
50256
- reviewScope: "adaptation_cohort",
50257
- recurrencePolicy: {
50258
- minimumDistinctAdaptationTasks: 2,
50259
- primaryTurnIsAnchorOnly: true
50260
- },
50261
- attemptCount: attempts.length,
50262
- passedCount: attempts.filter((attempt) => attempt.passed).length,
50263
- scoredAttemptCount: attempts.filter(
50264
- (attempt) => typeof attempt.score === "number"
50265
- ).length,
50266
- tasksWithToolFailures: attempts.filter(
50267
- (attempt) => attempt.toolFailureCount > 0
50268
- ).length,
50269
- totalToolFailureCount: attempts.reduce(
50270
- (total, attempt) => total + attempt.toolFailureCount,
50271
- 0
50272
- ),
50273
- totalLatencyMs: attempts.reduce(
50274
- (total, attempt) => total + attempt.latencyMs,
50275
- 0
50276
- ),
50277
- totalTokens: attempts.reduce(
50278
- (total, attempt) => total + attempt.usage.totalTokens,
50279
- 0
50280
- ),
50281
- behaviorFamilies,
50282
- crossTaskToolFailureGroups,
50283
- primaryEvidenceAnchor: selectPrimaryEvidenceAnchor(
50284
- attempts,
50285
- crossTaskToolFailureGroups
50286
- ),
50287
- attempts
50288
- };
50289
- }
50290
-
50291
50073
  // ../server/src/training/harness-refiner-benchmark-refiner-stage.ts
50292
- async function runHarnessRefinerBenchmarkRefinerStage(input) {
50293
- const refinerBoundaries = [];
50294
- const refinerUsage = emptyUsageCategory();
50295
- for (const result of input.adaptationAttempts) {
50296
- const attempt = result.attempt;
50297
- const sessionId = stringMetadata2(attempt.metadata, "sessionId");
50298
- const turnId = stringMetadata2(attempt.metadata, "turnId");
50299
- if (!sessionId || !turnId) continue;
50300
- const session = await input.store.getSession(sessionId);
50301
- if (!session) continue;
50302
- const overlay = await ensureLocalHarnessRunOverlay({
50303
- store: input.store,
50304
- runId: session.id,
50305
- workspace: input.baselineRuntime.workspace,
50306
- harnessRelease: {
50307
- id: input.baselineRuntime.release.harnessRelease.id,
50308
- contentHash: input.baselineRuntime.release.harnessRelease.contentHash
50309
- },
50310
- admittedAt: attempt.startedAt
50311
- });
50312
- const task = input.taskset.tasks.find((candidate2) => candidate2.id === attempt.taskId);
50313
- if (!task) continue;
50314
- const turn = TurnSchema.parse({
50315
- id: turnId,
50316
- sessionId: session.id,
50317
- providerTurnId: null,
50318
- modelRef: input.model,
50319
- prompt: taskPrompt2(task),
50320
- startedAt: attempt.startedAt,
50321
- completedAt: attempt.completedAt,
50322
- status: "completed",
50323
- error: null,
50324
- metadata: {
50325
- automatedTasksetWorkAttempt: true,
50326
- benchmarkId: "harness-refiner",
50327
- modelRunId: input.modelRun.id,
50328
- attemptId: attempt.id
50329
- },
50330
- harnessSnapshot: {
50331
- schemaVersion: "openpond.harnessTurnSnapshot.v1",
50332
- workspaceId: input.baselineRuntime.workspace.id,
50333
- workspaceRevision: input.baselineRuntime.workspace.revision,
50334
- sourceRevision: input.baselineRuntime.workspace.sourceRevision,
50335
- channelName: input.baselineRuntime.workspace.currentChannel.name,
50336
- channelRevision: input.baselineRuntime.workspace.currentChannel.revision,
50337
- harnessRelease: overlay.baseHarnessRelease,
50338
- overlay: {
50339
- id: overlay.id,
50340
- revision: overlay.revision,
50341
- contentHash: overlay.contentHash
50342
- }
50343
- }
50344
- });
50345
- if (!await input.store.getTurn(turn.id)) await input.store.insertTurn(turn);
50346
- const assistantOutput = attempt.output.text;
50347
- if (typeof assistantOutput === "string" && assistantOutput.trim()) {
50348
- await input.store.appendRuntimeEvent(event({
50349
- sessionId: session.id,
50350
- turnId: turn.id,
50351
- name: "assistant.delta",
50352
- source: "server",
50353
- appId: session.appId,
50354
- status: "completed",
50355
- output: assistantOutput
50356
- }));
50357
- }
50358
- const gradeEvidence = JSON.stringify({
50359
- schemaVersion: result.grade.schemaVersion,
50360
- id: result.grade.id,
50361
- passed: result.grade.passed,
50362
- score: result.grade.score,
50363
- failureClass: result.grade.failureClass,
50364
- rewardEligible: result.grade.rewardEligible,
50365
- feedback: result.grade.feedback,
50366
- evaluationCriteria: task.expectedOutput,
50367
- attempt: {
50368
- status: result.attempt.infrastructureError ? "infrastructure_failure" : "completed",
50369
- infrastructureError: result.attempt.infrastructureError,
50370
- outputPresent: typeof result.attempt.output.text === "string" && result.attempt.output.text.trim().length > 0,
50371
- artifactCount: result.artifacts.length,
50372
- runtimeEventCount: result.attempt.runtimeEventRefs.length,
50373
- latencyMs: result.attempt.latencyMs,
50374
- usage: attemptUsageSummary(result.attempt.metadata.usage)
50375
- }
50376
- });
50074
+ async function materializeBenchmarkRefinerBoundary(input) {
50075
+ const attempt = input.result.attempt;
50076
+ const sessionId = stringMetadata2(attempt.metadata, "sessionId");
50077
+ const turnId = stringMetadata2(attempt.metadata, "turnId");
50078
+ if (!sessionId || !turnId) return null;
50079
+ const session = await input.store.getSession(sessionId);
50080
+ if (!session) return null;
50081
+ const overlay = await ensureLocalHarnessRunOverlay({
50082
+ store: input.store,
50083
+ runId: session.id,
50084
+ workspace: input.runtime.workspace,
50085
+ harnessRelease: {
50086
+ id: input.runtime.release.harnessRelease.id,
50087
+ contentHash: input.runtime.release.harnessRelease.contentHash
50088
+ },
50089
+ admittedAt: attempt.startedAt
50090
+ });
50091
+ const task = input.taskset.tasks.find((candidate2) => candidate2.id === attempt.taskId);
50092
+ if (!task) return null;
50093
+ const turn = TurnSchema.parse({
50094
+ id: turnId,
50095
+ sessionId: session.id,
50096
+ providerTurnId: null,
50097
+ modelRef: input.model,
50098
+ prompt: taskPrompt2(task),
50099
+ startedAt: attempt.startedAt,
50100
+ completedAt: attempt.completedAt,
50101
+ status: "completed",
50102
+ error: null,
50103
+ metadata: {
50104
+ automatedTasksetWorkAttempt: true,
50105
+ benchmarkId: "harness-refiner",
50106
+ modelRunId: input.modelRun.id,
50107
+ attemptId: attempt.id
50108
+ },
50109
+ harnessSnapshot: {
50110
+ schemaVersion: "openpond.harnessTurnSnapshot.v1",
50111
+ workspaceId: input.runtime.workspace.id,
50112
+ workspaceRevision: input.runtime.workspace.revision,
50113
+ sourceRevision: input.runtime.workspace.sourceRevision,
50114
+ channelName: input.runtime.workspace.currentChannel.name,
50115
+ channelRevision: input.runtime.workspace.currentChannel.revision,
50116
+ harnessRelease: overlay.baseHarnessRelease,
50117
+ overlay: {
50118
+ id: overlay.id,
50119
+ revision: overlay.revision,
50120
+ contentHash: overlay.contentHash
50121
+ }
50122
+ }
50123
+ });
50124
+ if (!await input.store.getTurn(turn.id)) await input.store.insertTurn(turn);
50125
+ const assistantOutput = attempt.output.text;
50126
+ if (typeof assistantOutput === "string" && assistantOutput.trim()) {
50377
50127
  await input.store.appendRuntimeEvent(event({
50378
50128
  sessionId: session.id,
50379
50129
  turnId: turn.id,
50380
- name: "diagnostic",
50130
+ name: "assistant.delta",
50381
50131
  source: "server",
50382
50132
  appId: session.appId,
50383
- action: "taskset_grade",
50384
- status: result.grade.passed ? "completed" : "failed",
50385
- output: gradeEvidence,
50386
- error: result.grade.passed ? void 0 : gradeEvidence,
50387
- data: {
50388
- result: {
50389
- output: gradeEvidence,
50390
- passed: result.grade.passed,
50391
- score: result.grade.score
50392
- }
50393
- }
50133
+ status: "completed",
50134
+ output: assistantOutput
50394
50135
  }));
50395
- refinerBoundaries.push({ session, turn, result });
50396
50136
  }
50397
- const cohortEvidence = await buildHarnessRefinerBenchmarkCohortEvidence({
50398
- taskset: input.taskset,
50399
- adaptationAttempts: input.adaptationAttempts
50400
- });
50401
- const primaryBoundary = refinerBoundaries.find(
50402
- (boundary) => boundary.result.attempt.id === cohortEvidence.primaryEvidenceAnchor.attemptId
50403
- );
50404
- if (!primaryBoundary) throw new Error("Adaptation cohort produced no Refiner evidence.");
50137
+ const gradeEvidence = JSON.stringify({
50138
+ schemaVersion: input.result.grade.schemaVersion,
50139
+ id: input.result.grade.id,
50140
+ passed: input.result.grade.passed,
50141
+ score: input.result.grade.score,
50142
+ failureClass: input.result.grade.failureClass,
50143
+ rewardEligible: input.result.grade.rewardEligible,
50144
+ feedback: input.result.grade.feedback,
50145
+ evaluationCriteria: task.expectedOutput,
50146
+ attempt: {
50147
+ status: attempt.infrastructureError ? "infrastructure_failure" : "completed",
50148
+ infrastructureError: attempt.infrastructureError,
50149
+ outputPresent: typeof attempt.output.text === "string" && attempt.output.text.trim().length > 0,
50150
+ artifactCount: input.result.artifacts.length,
50151
+ runtimeEventCount: attempt.runtimeEventRefs.length,
50152
+ modelRequestCount: Array.isArray(attempt.metadata.usage) ? attempt.metadata.usage.length : attempt.metadata.usage ? 1 : 0,
50153
+ latencyMs: attempt.latencyMs,
50154
+ usage: attemptUsageSummary(attempt.metadata.usage)
50155
+ }
50156
+ });
50157
+ await input.store.appendRuntimeEvent(event({
50158
+ sessionId: session.id,
50159
+ turnId: turn.id,
50160
+ name: "diagnostic",
50161
+ source: "server",
50162
+ appId: session.appId,
50163
+ action: "taskset_grade",
50164
+ status: input.result.grade.passed ? "completed" : "failed",
50165
+ output: gradeEvidence,
50166
+ error: input.result.grade.passed ? void 0 : gradeEvidence,
50167
+ data: {
50168
+ result: {
50169
+ output: gradeEvidence,
50170
+ passed: input.result.grade.passed,
50171
+ score: input.result.grade.score
50172
+ }
50173
+ }
50174
+ }));
50175
+ return { session, turn, result: input.result };
50176
+ }
50177
+ async function runBenchmarkRefinerAfterAttempt(input) {
50178
+ const boundary = await materializeBenchmarkRefinerBoundary(input);
50179
+ if (!boundary) {
50180
+ throw new Error(`Adaptation attempt ${input.result.attempt.id} has no Refiner boundary.`);
50181
+ }
50405
50182
  const detection = await recordLocalHarnessImprovementBoundary({
50406
50183
  store: input.store,
50407
- session: primaryBoundary.session,
50408
- turn: primaryBoundary.turn,
50184
+ session: boundary.session,
50185
+ turn: boundary.turn,
50409
50186
  boundaryKind: "turn_completed",
50410
50187
  now: input.now
50411
50188
  });
50412
- if (!detection || detection.trigger.decision !== "queue_refiner") {
50413
- throw new Error("Adaptation cohort did not produce one Refiner trigger.");
50189
+ if (!detection) {
50190
+ throw new Error(`Adaptation attempt ${input.result.attempt.id} produced no Refiner detection.`);
50414
50191
  }
50415
- const refinerResults = [];
50416
- input.budget.assertAvailable(`Refiner cohort ${detection.trigger.id}`);
50417
- let refinerProviderInvoked = false;
50418
- let refinerResult;
50192
+ if (detection.trigger.decision !== "queue_refiner") {
50193
+ return { detection, result: null };
50194
+ }
50195
+ input.budget.assertAvailable(`Refiner turn ${detection.trigger.id}`);
50196
+ const usage = emptyUsageCategory();
50197
+ let providerInvoked = false;
50419
50198
  try {
50420
- refinerResult = await runLocalHarnessRefinerWorker({
50199
+ const result = await runLocalHarnessRefinerWorker({
50421
50200
  store: input.store,
50422
50201
  storeDir: input.storeDir,
50423
50202
  trigger: detection.trigger,
50424
- additionalEvidence: cohortEvidence,
50425
50203
  stream: async function* (streamInput) {
50426
- refinerProviderInvoked = true;
50204
+ providerInvoked = true;
50427
50205
  for await (const delta of input.refinerStream({
50428
50206
  ...streamInput,
50429
50207
  model: input.model,
50430
50208
  pricing: input.admittedPricing
50431
50209
  })) {
50432
- if (delta.usage !== void 0) {
50433
- addUsage(refinerUsage, delta.usage, delta.costUsd);
50434
- }
50210
+ if (delta.usage !== void 0) addUsage(usage, delta.usage, delta.costUsd);
50435
50211
  if (delta.text) yield { text: delta.text };
50436
50212
  }
50437
50213
  },
50438
50214
  signal: input.signal,
50439
50215
  now: input.now
50440
50216
  });
50217
+ return { detection, result };
50441
50218
  } finally {
50442
- if (refinerProviderInvoked) {
50443
- input.budget.charge(refinerUsage.costUsd, "Refiner");
50219
+ if (providerInvoked) {
50220
+ input.budget.charge(usage.costUsd, "Refiner");
50444
50221
  await checkpointRefinerUsage(
50445
50222
  input.store,
50446
50223
  input.modelRun.id,
50447
- refinerUsage,
50224
+ usage,
50448
50225
  input.budget.observedSpendUsd
50449
50226
  );
50450
50227
  }
50451
50228
  }
50452
- refinerResults.push(refinerResult);
50453
- const candidateWorkspace = await input.store.getHarnessWorkspace(input.baselineRuntime.workspace.id);
50454
- if (!candidateWorkspace?.currentChannel.release) {
50455
- throw new Error("Isolated Harness candidate is unavailable.");
50456
- }
50457
- const candidateRecord = await input.store.getHarnessReleaseRecord(
50458
- candidateWorkspace.currentChannel.release.contentHash
50459
- );
50460
- if (!candidateRecord) throw new Error("Isolated Harness candidate release is unavailable.");
50461
- const candidateRuntime = await loadLocalHarnessRuntimeFromRelease({
50462
- workspace: candidateWorkspace,
50463
- release: candidateRecord
50464
- });
50229
+ }
50230
+
50231
+ // ../server/src/training/harness-refiner-benchmark-sequential-stage.ts
50232
+ async function runSequentialHarnessAdaptation(input) {
50233
+ const runRefinerAfterAttempt = input.adapters?.runRefinerAfterAttempt ?? runBenchmarkRefinerAfterAttempt;
50234
+ const loadRuntimeFromRelease = input.adapters?.loadRuntimeFromRelease ?? loadLocalHarnessRuntimeFromRelease;
50235
+ const buildLineage = input.adapters?.buildLineage ?? benchmarkLineage;
50465
50236
  await input.store.setHarnessBackgroundReviewSettings({
50466
- workspaceId: input.baselineRuntime.workspace.id,
50467
- enabled: false,
50237
+ workspaceId: input.initialRuntime.workspace.id,
50238
+ enabled: true,
50468
50239
  updatedAt: input.now()
50469
50240
  });
50241
+ let runtime = input.initialRuntime;
50242
+ const attempts = [];
50243
+ const refinerResults = [];
50244
+ const steps = [];
50245
+ try {
50246
+ for (const [ordinal, taskId] of input.taskIds.entries()) {
50247
+ input.signal.throwIfAborted();
50248
+ const before = {
50249
+ id: runtime.release.harnessRelease.id,
50250
+ contentHash: runtime.release.harnessRelease.contentHash
50251
+ };
50252
+ const executed = await input.evaluation.execute({
50253
+ tasksetId: input.taskset.id,
50254
+ taskId,
50255
+ model: input.model,
50256
+ reasoningEffort: input.reasoningEffort,
50257
+ seed: input.seed,
50258
+ attempt: 0,
50259
+ sampling: { maxOutputTokens: 4096, temperature: 0, topP: 1 },
50260
+ releasedHarness: releasedHarness(runtime.release, runtime.instructionContext),
50261
+ hostedTokenPricing: input.admittedPricing,
50262
+ parentModelRunId: input.modelRun.id,
50263
+ signal: input.signal,
50264
+ toolEvidence: frozenToolEvidence(input.evidenceSnapshot, "replay", "adaptation")
50265
+ });
50266
+ await input.onAttemptComplete(executed);
50267
+ const evidence = {
50268
+ attempt: executed.attempt,
50269
+ grade: executed.grade,
50270
+ artifacts: executed.artifacts,
50271
+ receiptContentHash: executed.portable.receipt.contentHash
50272
+ };
50273
+ attempts.push(evidence);
50274
+ const refined = await runRefinerAfterAttempt({
50275
+ store: input.store,
50276
+ storeDir: input.storeDir,
50277
+ modelRun: input.modelRun,
50278
+ model: input.model,
50279
+ taskset: input.taskset,
50280
+ runtime,
50281
+ result: evidence,
50282
+ budget: input.budget,
50283
+ admittedPricing: input.admittedPricing,
50284
+ refinerStream: input.refinerStream,
50285
+ signal: input.signal,
50286
+ now: input.now
50287
+ });
50288
+ if (refined.result) refinerResults.push(refined.result);
50289
+ const workspace = await input.store.getHarnessWorkspace(runtime.workspace.id);
50290
+ const releaseRef = workspace?.currentChannel.release;
50291
+ if (!workspace || !releaseRef) {
50292
+ throw new Error("Sequential Refiner did not retain a current Harness release.");
50293
+ }
50294
+ const release = await input.store.getHarnessReleaseRecord(releaseRef.contentHash);
50295
+ if (!release) throw new Error("Sequential Refiner release is unavailable.");
50296
+ runtime = await loadRuntimeFromRelease({ workspace, release });
50297
+ const after = {
50298
+ id: runtime.release.harnessRelease.id,
50299
+ contentHash: runtime.release.harnessRelease.contentHash
50300
+ };
50301
+ steps.push({
50302
+ ordinal,
50303
+ taskId,
50304
+ attemptId: evidence.attempt.id,
50305
+ inputHarness: before,
50306
+ outputHarness: after,
50307
+ trigger: {
50308
+ id: refined.detection.trigger.id,
50309
+ contentHash: refined.detection.trigger.contentHash,
50310
+ decision: refined.detection.trigger.decision
50311
+ },
50312
+ outcome: refined.result ? {
50313
+ id: refined.result.outcome.id,
50314
+ contentHash: refined.result.outcome.contentHash,
50315
+ decision: refined.result.outcome.decision
50316
+ } : null,
50317
+ changed: before.contentHash !== after.contentHash
50318
+ });
50319
+ }
50320
+ } finally {
50321
+ await input.store.setHarnessBackgroundReviewSettings({
50322
+ workspaceId: input.initialRuntime.workspace.id,
50323
+ enabled: false,
50324
+ updatedAt: input.now()
50325
+ });
50326
+ }
50327
+ const usage = attempts.reduce(
50328
+ (total, evidence) => {
50329
+ const records = Array.isArray(evidence.attempt.metadata.usage) ? evidence.attempt.metadata.usage : [evidence.attempt.metadata.usage];
50330
+ for (const value of records) {
50331
+ if (!value || typeof value !== "object" || Array.isArray(value)) continue;
50332
+ const record11 = value;
50333
+ const inputTokens = numericToken(record11, ["promptTokens", "inputTokens"]);
50334
+ const outputTokens = numericToken(record11, ["completionTokens", "outputTokens"]);
50335
+ total.inputTokens += inputTokens;
50336
+ total.outputTokens += outputTokens;
50337
+ total.totalTokens += numericToken(record11, ["totalTokens"]) || inputTokens + outputTokens;
50338
+ }
50339
+ return total;
50340
+ },
50341
+ { inputTokens: 0, outputTokens: 0, totalTokens: 0 }
50342
+ );
50343
+ const costs = attempts.flatMap(
50344
+ (evidence) => typeof evidence.attempt.costUsd === "number" ? [evidence.attempt.costUsd] : []
50345
+ );
50346
+ const initialHarness = {
50347
+ id: input.initialRuntime.release.harnessRelease.id,
50348
+ contentHash: input.initialRuntime.release.harnessRelease.contentHash
50349
+ };
50350
+ const finalHarness = {
50351
+ id: runtime.release.harnessRelease.id,
50352
+ contentHash: runtime.release.harnessRelease.contentHash
50353
+ };
50354
+ const summaryCore = {
50355
+ schemaVersion: "openpond.sequentialHarnessAdaptation.v1",
50356
+ id: `sequential-adaptation-${input.modelRun.id}`,
50357
+ initialHarness,
50358
+ finalHarness,
50359
+ attemptCount: attempts.length,
50360
+ passedCount: attempts.filter((evidence) => evidence.grade.passed).length,
50361
+ terminalCount: attempts.filter((evidence) => !evidence.attempt.infrastructureError).length,
50362
+ usage,
50363
+ costUsd: costs.length ? costs.reduce((total, cost) => total + cost, 0) : null,
50364
+ latencyMs: attempts.reduce((total, evidence) => total + evidence.attempt.latencyMs, 0),
50365
+ steps,
50366
+ createdAt: input.now()
50367
+ };
50368
+ const summary2 = {
50369
+ ...summaryCore,
50370
+ contentHash: contentHash(summaryCore)
50371
+ };
50470
50372
  const outcomes = await input.store.listHarnessImprovementArtifacts(
50471
- input.baselineRuntime.workspace.id,
50373
+ input.initialRuntime.workspace.id,
50472
50374
  "refiner_outcome",
50473
50375
  1e3
50474
50376
  );
@@ -50477,23 +50379,36 @@ async function runHarnessRefinerBenchmarkRefinerStage(input) {
50477
50379
  contentHash: contentHash(outcomes),
50478
50380
  outcomeCount: outcomes.length
50479
50381
  };
50480
- const lineage = await benchmarkLineage({
50382
+ const lineage = await buildLineage({
50481
50383
  store: input.store,
50482
- workspaceId: input.baselineRuntime.workspace.id,
50483
- adaptationAttempts: input.adaptationAttempts,
50384
+ workspaceId: input.initialRuntime.workspace.id,
50385
+ adaptationAttempts: attempts,
50484
50386
  refinerResults,
50485
- candidateRelease: candidateRecord.harnessRelease,
50486
- refinerInputHash: contentHash(cohortEvidence)
50387
+ candidateRelease: finalHarness,
50388
+ refinerInputHash: contentHash(attempts.map((evidence) => ({
50389
+ attempt: evidence.attempt.id,
50390
+ receipt: evidence.receiptContentHash,
50391
+ grade: contentHash(evidence.grade)
50392
+ })))
50487
50393
  });
50488
- const frozenEvidence = input.evidenceSnapshot.manifest();
50489
50394
  return {
50490
- candidateRecord,
50491
- candidateRuntime,
50395
+ attempts,
50396
+ runtime,
50397
+ summary: summary2,
50492
50398
  refinerStage,
50493
50399
  lineage,
50494
- frozenEvidence
50400
+ refinerResults
50495
50401
  };
50496
50402
  }
50403
+ function numericToken(record11, keys) {
50404
+ for (const key of keys) {
50405
+ const value = record11[key];
50406
+ if (typeof value === "number" && Number.isFinite(value) && value >= 0) {
50407
+ return Math.trunc(value);
50408
+ }
50409
+ }
50410
+ return 0;
50411
+ }
50497
50412
 
50498
50413
  // ../server/src/training/harness-refiner-benchmark-service.ts
50499
50414
  var BenchmarkRunCancelledError = class extends Error {
@@ -50758,7 +50673,6 @@ function createHarnessRefinerBenchmarkService(deps) {
50758
50673
  executionPlan,
50759
50674
  baselinePlan,
50760
50675
  adaptationPlan,
50761
- candidateAdaptationPlan,
50762
50676
  candidatePlan,
50763
50677
  admittedPricing,
50764
50678
  budget,
@@ -50859,11 +50773,6 @@ function createHarnessRefinerBenchmarkService(deps) {
50859
50773
  completedAttempts: completedBeforeStage(executionPlan, "adaptation"),
50860
50774
  totalAttempts
50861
50775
  });
50862
- await deps.store.setHarnessBackgroundReviewSettings({
50863
- workspaceId: isolated.workspace.id,
50864
- enabled: true,
50865
- updatedAt: now2()
50866
- });
50867
50776
  const executedAdaptation = await deps.evaluation.executeBenchmark({
50868
50777
  tasksetId: taskset.id,
50869
50778
  phase: "baseline",
@@ -50895,168 +50804,47 @@ function createHarnessRefinerBenchmarkService(deps) {
50895
50804
  totalAttempts
50896
50805
  });
50897
50806
  }
50898
- if (resumeFromRefiner) {
50899
- const priorOutcomes = await deps.store.listHarnessImprovementArtifacts(
50900
- isolated.workspace.id,
50901
- "refiner_outcome",
50902
- 1e3
50807
+ if (resumeFromRefiner && !modelRun.evaluationProgress?.evidenceSnapshot) {
50808
+ throw new Error(
50809
+ "Sequential adaptation cannot resume because the interrupted run did not preserve its exact frozen evidence snapshot. Start a new run to preserve a valid causal comparison."
50903
50810
  );
50904
- const currentWorkspace = await deps.store.getHarnessWorkspace(isolated.workspace.id);
50905
- const currentReleaseRef = currentWorkspace?.currentChannel.release;
50906
- if (priorOutcomes.length > 0 && currentReleaseRef && currentReleaseRef.contentHash === baseline.run.harnessRelease.contentHash) {
50907
- await deps.store.setHarnessBackgroundReviewSettings({
50908
- workspaceId: isolated.workspace.id,
50909
- enabled: false,
50910
- updatedAt: now2()
50911
- });
50912
- const latestRun2 = await deps.store.getModelRun(modelRun.id) ?? modelRun;
50913
- const accounting = latestRun2.evaluationProgress?.accounting ?? emptyEvaluationAccounting();
50914
- const frozenEvidence2 = evidenceSnapshot.manifest();
50915
- const refinerStage2 = {
50916
- id: `benchmark-refiner-${modelRun.id}`,
50917
- contentHash: contentHash(priorOutcomes),
50918
- outcomeCount: priorOutcomes.length
50919
- };
50920
- const receiptCore2 = {
50921
- schemaVersion: "openpond.modelEvaluationStopReceipt.v1",
50922
- benchmarkId: "harness-refiner",
50923
- terminalClassification: "inconclusive",
50924
- stopReason: "candidate_harness_unchanged",
50925
- reason: "Refiner produced no changed Harness candidate. Candidate replay was skipped because no effect can be attributed to refinement.",
50926
- stoppedAfter: "refiner",
50927
- baselineHarness: baseline.run.harnessRelease,
50928
- candidateHarness: currentReleaseRef,
50929
- refiner: refinerStage2,
50930
- usage: accounting.usage,
50931
- budget: {
50932
- maximumSpendUsd: budget.maximumSpendUsd,
50933
- observedSpendUsd: accounting.observedSpendUsd,
50934
- enforced: true
50935
- },
50936
- evidenceSnapshot: {
50937
- id: frozenEvidence2.id,
50938
- contentHash: frozenEvidence2.contentHash
50939
- },
50940
- attempts: accounting.attempts
50941
- };
50942
- const receipt2 = ModelEvaluationStopReceiptSchema.parse({
50943
- ...receiptCore2,
50944
- contentHash: contentHash(receiptCore2)
50945
- });
50946
- const completedAt2 = now2();
50947
- return deps.store.saveModelRun(ModelRunSchema.parse({
50948
- ...latestRun2,
50949
- status: "succeeded",
50950
- receipt: receipt2,
50951
- failure: null,
50952
- completedAt: completedAt2,
50953
- updatedAt: completedAt2
50954
- }));
50955
- }
50956
50811
  }
50957
- const {
50958
- candidateRecord,
50959
- candidateRuntime,
50960
- refinerStage,
50961
- lineage,
50962
- frozenEvidence
50963
- } = await runHarnessRefinerBenchmarkRefinerStage({
50812
+ await updateProgress(deps.store, modelRun.id, {
50813
+ stage: "candidate_adaptation",
50814
+ completedAttempts: completedBeforeStage(executionPlan, "candidate_adaptation"),
50815
+ totalAttempts
50816
+ });
50817
+ const candidateAdaptation = await runSequentialHarnessAdaptation({
50964
50818
  store: deps.store,
50965
50819
  storeDir: deps.storeDir,
50820
+ evaluation: deps.evaluation,
50966
50821
  modelRun,
50967
50822
  model: input.model,
50823
+ reasoningEffort: input.reasoningEffort,
50968
50824
  taskset,
50969
- baselineRuntime,
50970
- adaptationAttempts,
50825
+ taskIds: candidateAdaptationPlan.taskIds,
50826
+ seed: input.seeds[0],
50827
+ initialRuntime: baselineRuntime,
50971
50828
  evidenceSnapshot,
50972
50829
  budget,
50973
50830
  admittedPricing,
50974
50831
  refinerStream: deps.refinerStream,
50975
50832
  signal: context.signal,
50833
+ onAttemptComplete: observeStageAttempts(
50834
+ "candidate_adaptation",
50835
+ "sequential adaptation treatment"
50836
+ ),
50976
50837
  now: now2
50977
50838
  });
50978
- const harnessChanged = baseline.run.harnessRelease.contentHash !== candidateRecord.harnessRelease.contentHash;
50979
- if (!harnessChanged || !lineage.valid) {
50980
- const stopReason = harnessChanged ? "candidate_lineage_invalid" : "candidate_harness_unchanged";
50981
- const reason = harnessChanged ? "Refiner produced a candidate whose causal lineage did not validate. Candidate replay was skipped." : "Refiner produced no changed Harness candidate. Candidate replay was skipped because no effect can be attributed to refinement.";
50982
- const latestRun2 = await deps.store.getModelRun(modelRun.id) ?? modelRun;
50983
- const accounting = latestRun2.evaluationProgress?.accounting ?? emptyEvaluationAccounting();
50984
- const receiptCore2 = {
50985
- schemaVersion: "openpond.modelEvaluationStopReceipt.v1",
50986
- benchmarkId: "harness-refiner",
50987
- terminalClassification: "inconclusive",
50988
- stopReason,
50989
- reason,
50990
- stoppedAfter: "refiner",
50991
- baselineHarness: baseline.run.harnessRelease,
50992
- candidateHarness: {
50993
- id: candidateRecord.harnessRelease.id,
50994
- contentHash: candidateRecord.harnessRelease.contentHash
50995
- },
50996
- refiner: refinerStage,
50997
- usage: accounting.usage,
50998
- budget: {
50999
- maximumSpendUsd: budget.maximumSpendUsd,
51000
- observedSpendUsd: budget.observedSpendUsd,
51001
- enforced: true
51002
- },
51003
- evidenceSnapshot: {
51004
- id: frozenEvidence.id,
51005
- contentHash: frozenEvidence.contentHash
51006
- },
51007
- attempts: accounting.attempts
51008
- };
51009
- const receipt2 = ModelEvaluationStopReceiptSchema.parse({
51010
- ...receiptCore2,
51011
- contentHash: contentHash(receiptCore2)
51012
- });
51013
- const completedAt2 = now2();
51014
- return deps.store.saveModelRun(ModelRunSchema.parse({
51015
- ...latestRun2,
51016
- status: "succeeded",
51017
- receipt: receipt2,
51018
- failure: null,
51019
- completedAt: completedAt2,
51020
- updatedAt: completedAt2
51021
- }));
51022
- }
51023
- if (resumeFromRefiner && !modelRun.evaluationProgress?.evidenceSnapshot) {
51024
- throw new Error(
51025
- "Candidate replay cannot resume because the interrupted run did not preserve its exact frozen evidence snapshot. Start a new run to preserve a valid causal comparison."
51026
- );
51027
- }
51028
- await updateProgress(deps.store, modelRun.id, {
51029
- stage: "candidate_adaptation",
51030
- completedAttempts: completedBeforeStage(executionPlan, "candidate_adaptation"),
51031
- totalAttempts
51032
- });
50839
+ const candidateRuntime = candidateAdaptation.runtime;
50840
+ const refinerStage = candidateAdaptation.refinerStage;
50841
+ const lineage = candidateAdaptation.lineage;
50842
+ const frozenEvidence = evidenceSnapshot.manifest();
50843
+ const harnessChanged = baseline.run.harnessRelease.contentHash !== candidateAdaptation.summary.finalHarness.contentHash;
51033
50844
  const candidateHarness = releasedHarness(
51034
50845
  candidateRuntime.release,
51035
50846
  candidateRuntime.instructionContext
51036
50847
  );
51037
- const candidateAdaptation = await deps.evaluation.executeBenchmark({
51038
- tasksetId: taskset.id,
51039
- phase: "candidate",
51040
- model: input.model,
51041
- reasoningEffort: input.reasoningEffort,
51042
- seeds: input.seeds,
51043
- repetitions: input.repetitions,
51044
- split: candidateAdaptationPlan.split,
51045
- taskIds: candidateAdaptationPlan.taskIds,
51046
- sampling: { maxOutputTokens: 4096, temperature: 0, topP: 1 },
51047
- releasedHarness: candidateHarness,
51048
- hostedTokenPricing: admittedPricing,
51049
- parentModelRunId: modelRun.id,
51050
- signal: context.signal,
51051
- toolEvidence: frozenToolEvidence(evidenceSnapshot, "replay", "adaptation"),
51052
- onAttemptComplete: observeStageAttempts(
51053
- "candidate_adaptation",
51054
- "candidate adaptation replay"
51055
- )
51056
- });
51057
- if (!candidateAdaptation.comparison) {
51058
- throw new Error("Adaptation replay comparison was not produced.");
51059
- }
51060
50848
  await updateProgress(deps.store, modelRun.id, {
51061
50849
  stage: "candidate",
51062
50850
  completedAttempts: completedBeforeStage(executionPlan, "candidate"),
@@ -51093,7 +50881,7 @@ function createHarnessRefinerBenchmarkService(deps) {
51093
50881
  baseline: baseline.run,
51094
50882
  adaptation: adaptation.run,
51095
50883
  refiner: refinerStage,
51096
- candidateAdaptation: candidateAdaptation.run,
50884
+ candidateAdaptation: candidateAdaptation.summary,
51097
50885
  candidate: candidate2.run,
51098
50886
  comparison: candidate2.comparison,
51099
50887
  executionPlan,
@@ -51128,17 +50916,17 @@ function createHarnessRefinerBenchmarkService(deps) {
51128
50916
  baseline: baseline.run,
51129
50917
  adaptation: adaptation.run,
51130
50918
  candidate: candidate2.run,
51131
- harnessChanged: baseline.run.harnessRelease.contentHash !== candidate2.run.harnessRelease.contentHash,
51132
- candidateAdaptation: candidateAdaptation.run,
50919
+ harnessChanged,
50920
+ candidateAdaptation: candidateAdaptation.summary,
51133
50921
  lineageValid: lineage.valid,
51134
50922
  infrastructureValid
51135
50923
  });
51136
50924
  const invalidReasons = comparisonInvalidReasons({
51137
50925
  baseline: baseline.run,
51138
50926
  adaptation: adaptation.run,
51139
- candidateAdaptation: candidateAdaptation.run,
50927
+ candidateAdaptation: candidateAdaptation.summary,
51140
50928
  candidate: candidate2.run,
51141
- harnessChanged: baseline.run.harnessRelease.contentHash !== candidate2.run.harnessRelease.contentHash,
50929
+ harnessChanged,
51142
50930
  lineageValid: lineage.valid,
51143
50931
  infrastructureValid
51144
50932
  });
@@ -51169,8 +50957,8 @@ function createHarnessRefinerBenchmarkService(deps) {
51169
50957
  contentHash: adaptation.run.contentHash
51170
50958
  },
51171
50959
  candidateAdaptation: {
51172
- id: candidateAdaptation.run.id,
51173
- contentHash: candidateAdaptation.run.contentHash
50960
+ id: candidateAdaptation.summary.id,
50961
+ contentHash: candidateAdaptation.summary.contentHash
51174
50962
  },
51175
50963
  refiner: { id: refinerStage.id, contentHash: refinerStage.contentHash },
51176
50964
  candidate: { id: candidate2.run.id, contentHash: candidate2.run.contentHash },
@@ -51184,10 +50972,10 @@ function createHarnessRefinerBenchmarkService(deps) {
51184
50972
  baselinePassRate: candidate2.comparison.baselinePassRate,
51185
50973
  candidatePassRate: candidate2.comparison.candidatePassRate,
51186
50974
  adaptationBaselinePassRate: adaptation.run.passedCount / adaptation.run.attemptCount,
51187
- adaptationCandidatePassRate: candidateAdaptation.run.passedCount / candidateAdaptation.run.attemptCount,
51188
- adaptationCandidatePassed: candidateAdaptation.run.passedCount === candidateAdaptation.run.attemptCount,
50975
+ adaptationCandidatePassRate: candidateAdaptation.summary.passedCount / candidateAdaptation.summary.attemptCount,
50976
+ adaptationCandidatePassed: candidateAdaptation.summary.passedCount === candidateAdaptation.summary.attemptCount,
51189
50977
  heldOutCandidatePassed: candidate2.run.passedCount === candidate2.run.attemptCount,
51190
- passed: candidate2.comparison.qualityPassed && candidateAdaptation.run.passedCount === candidateAdaptation.run.attemptCount && candidate2.run.passedCount === candidate2.run.attemptCount && infrastructureValid && terminalClassification !== "infrastructure_failure"
50978
+ passed: candidate2.comparison.qualityPassed && candidateAdaptation.summary.passedCount === candidateAdaptation.summary.attemptCount && candidate2.run.passedCount === candidate2.run.attemptCount && infrastructureValid && terminalClassification !== "infrastructure_failure"
51191
50979
  },
51192
50980
  foregroundTokenDelta: candidate2.comparison.foregroundTokenDelta,
51193
50981
  foregroundTokenDeltaPercent: candidate2.comparison.foregroundTokenDeltaPercent,
@@ -51336,14 +51124,14 @@ async function loadTaskAttemptGraderEvidence(input) {
51336
51124
  return {
51337
51125
  ...base,
51338
51126
  extraction: "pdf_text",
51339
- content: boundedText2(stdout)
51127
+ content: boundedText(stdout)
51340
51128
  };
51341
51129
  }
51342
51130
  if (isTextArtifact(artifact.mediaType)) {
51343
51131
  return {
51344
51132
  ...base,
51345
51133
  extraction: "text",
51346
- content: boundedText2(await readFile15(artifact.path, "utf8"))
51134
+ content: boundedText(await readFile15(artifact.path, "utf8"))
51347
51135
  };
51348
51136
  }
51349
51137
  return { ...base, extraction: "metadata_only" };
@@ -51361,7 +51149,7 @@ function isTextArtifact(mediaType) {
51361
51149
  mediaType?.startsWith("text/") || mediaType === "application/json" || mediaType === "application/xml"
51362
51150
  );
51363
51151
  }
51364
- function boundedText2(value) {
51152
+ function boundedText(value) {
51365
51153
  return value.length <= MAX_EXTRACTED_CHARS ? value : `${value.slice(0, MAX_EXTRACTED_CHARS)}
51366
51154
  [truncated]`;
51367
51155
  }
@@ -51421,7 +51209,7 @@ function usageCost(usage, pricing) {
51421
51209
  }
51422
51210
 
51423
51211
  // ../server/src/training/benchmark-tasksets.ts
51424
- import { createHash as createHash16 } from "node:crypto";
51212
+ import { createHash as createHash15 } from "node:crypto";
51425
51213
  import { mkdir as mkdir11, readFile as readFile16, writeFile as writeFile7 } from "node:fs/promises";
51426
51214
  import path58 from "node:path";
51427
51215
  var BUILTIN_DEFINITION_ID = "harness-refiner";
@@ -51474,8 +51262,8 @@ function definitionForRelease(release) {
51474
51262
  return createBenchmarkDefinition({
51475
51263
  schemaVersion: "openpond.benchmarkDefinition.v1",
51476
51264
  id: BUILTIN_DEFINITION_ID,
51477
- title: "Harness Refiner",
51478
- description: "Measures whether evidence-driven Harness improvements preserve quality while reducing held-out foreground tokens.",
51265
+ title: "Harness Refiner 08112026",
51266
+ description: "Measures whether sequential, evidence-driven Harness improvements preserve quality while reducing future-task foreground tokens.",
51479
51267
  tasksetRelease: { id: release.id, contentHash: release.contentHash },
51480
51268
  adaptationSplit: "validation",
51481
51269
  evaluationSplit: "frozen_eval",
@@ -51485,7 +51273,7 @@ function definitionForRelease(release) {
51485
51273
  adaptation: release.tasks.filter((task) => task.split === "validation").length,
51486
51274
  evaluation: release.tasks.filter((task) => task.split === "frozen_eval").length
51487
51275
  },
51488
- metadata: { builtin: true }
51276
+ metadata: { builtin: true, protocol: "sequential_product_lifecycle" }
51489
51277
  });
51490
51278
  }
51491
51279
  async function projectRelease(input) {
@@ -51516,7 +51304,7 @@ async function projectRelease(input) {
51516
51304
  managedAssetById.set(asset.id, {
51517
51305
  artifactRef,
51518
51306
  fileName,
51519
- sha256: createHash16("sha256").update(contents).digest("hex"),
51307
+ sha256: createHash15("sha256").update(contents).digest("hex"),
51520
51308
  sizeBytes: Buffer.byteLength(contents),
51521
51309
  mediaType: asset.mediaType
51522
51310
  });
@@ -51809,7 +51597,7 @@ function stringValue11(value) {
51809
51597
 
51810
51598
  // ../server/src/training/local-taskset-work-runtime.ts
51811
51599
  import { execFile as execFile8 } from "node:child_process";
51812
- import { createHash as createHash17, randomUUID as randomUUID24 } from "node:crypto";
51600
+ import { createHash as createHash16, randomUUID as randomUUID24 } from "node:crypto";
51813
51601
  import { existsSync as existsSync12, promises as fs25 } from "node:fs";
51814
51602
  import os9 from "node:os";
51815
51603
  import path59 from "node:path";
@@ -51976,11 +51764,11 @@ async function executeLocalWorkspaceTool(input) {
51976
51764
  await fs25.writeFile(destination, bytes2, { mode: 384 });
51977
51765
  const outputRef = FileOutputRefSchema.parse({
51978
51766
  kind: "file",
51979
- id: `output-${createHash17("sha256").update(destination).digest("hex").slice(0, 24)}`,
51767
+ id: `output-${createHash16("sha256").update(destination).digest("hex").slice(0, 24)}`,
51980
51768
  title,
51981
51769
  contentType: contentType2,
51982
51770
  sizeBytes: bytes2.byteLength,
51983
- sha256: createHash17("sha256").update(bytes2).digest("hex"),
51771
+ sha256: createHash16("sha256").update(bytes2).digest("hex"),
51984
51772
  sourceTaskId: input.session.id,
51985
51773
  sourceTurnId: input.turnId ?? `local-work-${input.session.id}`,
51986
51774
  revision: 1,
@@ -52105,7 +51893,7 @@ function runCommand5(command, cwd, timeoutMs) {
52105
51893
  function localStatus(root, state = "running") {
52106
51894
  return {
52107
51895
  sandbox: {
52108
- id: `desktop-local:${createHash17("sha256").update(root).digest("hex").slice(0, 16)}`,
51896
+ id: `desktop-local:${createHash16("sha256").update(root).digest("hex").slice(0, 16)}`,
52109
51897
  state,
52110
51898
  provider: "desktop-local"
52111
51899
  },
@@ -52117,7 +51905,7 @@ function fileData(target, bytes2) {
52117
51905
  return {
52118
51906
  path: target,
52119
51907
  sizeBytes: bytes2.byteLength,
52120
- sha256: createHash17("sha256").update(bytes2).digest("hex")
51908
+ sha256: createHash16("sha256").update(bytes2).digest("hex")
52121
51909
  };
52122
51910
  }
52123
51911
  function validationEvidence(value) {