@tangle-network/agent-eval 0.131.1 → 0.132.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (103) hide show
  1. package/CHANGELOG.md +7 -0
  2. package/dist/analyst/index.d.ts +5 -5
  3. package/dist/analyst/index.js +3 -3
  4. package/dist/{analyze-runs-CBnCgfse.d.ts → analyze-runs-AFDI5RI0.d.ts} +5 -5
  5. package/dist/{analyze-runs-CBnCgfse.d.ts.map → analyze-runs-AFDI5RI0.d.ts.map} +1 -1
  6. package/dist/belief-state/index.d.ts +4 -4
  7. package/dist/benchmarks/index.d.ts +1 -1
  8. package/dist/benchmarks/index.js +1 -1
  9. package/dist/{benchmarks-BDtRiOvE.js → benchmarks-DTrT3UH-.js} +3 -3
  10. package/dist/{benchmarks-BDtRiOvE.js.map → benchmarks-DTrT3UH-.js.map} +1 -1
  11. package/dist/campaign/index.d.ts +4 -4
  12. package/dist/campaign/index.js +2 -2
  13. package/dist/{campaign-CLG6y9jF.js → campaign-Cx6CfMR4.js} +4 -4
  14. package/dist/{campaign-CLG6y9jF.js.map → campaign-Cx6CfMR4.js.map} +1 -1
  15. package/dist/cli.js +1 -1
  16. package/dist/{client-w90OcvvR.d.ts → client-aZDHJiKO.d.ts} +4 -4
  17. package/dist/{client-w90OcvvR.d.ts.map → client-aZDHJiKO.d.ts.map} +1 -1
  18. package/dist/{code-agent-session-aYa3SGKz.d.ts → code-agent-session-D5URqc3_.d.ts} +2 -2
  19. package/dist/{code-agent-session-aYa3SGKz.d.ts.map → code-agent-session-D5URqc3_.d.ts.map} +1 -1
  20. package/dist/contract/index.d.ts +49 -22
  21. package/dist/contract/index.d.ts.map +1 -1
  22. package/dist/contract/index.js +275 -61
  23. package/dist/contract/index.js.map +1 -1
  24. package/dist/control.d.ts +1 -1
  25. package/dist/{cost-ledger-DIgQUFZZ.js → cost-ledger-ZAa_P4r0.js} +41 -7
  26. package/dist/cost-ledger-ZAa_P4r0.js.map +1 -0
  27. package/dist/{cost-ledger-Dye6jCgg.d.ts → cost-ledger-fGS_u_O1.d.ts} +18 -2
  28. package/dist/cost-ledger-fGS_u_O1.d.ts.map +1 -0
  29. package/dist/{default-registry-C-vFCSEc.js → default-registry-B1JcpnRv.js} +4 -13
  30. package/dist/default-registry-B1JcpnRv.js.map +1 -0
  31. package/dist/{default-registry-Cj1oUpLN.d.ts → default-registry-Cl3pHo4n.d.ts} +4 -4
  32. package/dist/{default-registry-Cj1oUpLN.d.ts.map → default-registry-Cl3pHo4n.d.ts.map} +1 -1
  33. package/dist/{eval-campaign-C2k-m4aY.js → eval-campaign-mDKhkdUq.js} +2 -2
  34. package/dist/{eval-campaign-C2k-m4aY.js.map → eval-campaign-mDKhkdUq.js.map} +1 -1
  35. package/dist/fuzz.d.ts +1 -1
  36. package/dist/fuzz.js +1 -1
  37. package/dist/hosted/index.d.ts +2 -2
  38. package/dist/{index-B11XCkdf2.d.ts → index-3cdlURSk2.d.ts} +2 -2
  39. package/dist/{index-B11XCkdf2.d.ts.map → index-3cdlURSk2.d.ts.map} +1 -1
  40. package/dist/{index-DNXoNaFF.d.ts → index-C2fkZhv_.d.ts} +2 -2
  41. package/dist/{index-DNXoNaFF.d.ts.map → index-C2fkZhv_.d.ts.map} +1 -1
  42. package/dist/{index-D_F6VAKe.d.ts → index-CXs7QlR5.d.ts} +3 -3
  43. package/dist/{index-D_F6VAKe.d.ts.map → index-CXs7QlR5.d.ts.map} +1 -1
  44. package/dist/{index-VTypFU3t.d.ts → index-FpfWFsKm.d.ts} +7 -7
  45. package/dist/{index-VTypFU3t.d.ts.map → index-FpfWFsKm.d.ts.map} +1 -1
  46. package/dist/{index-NPeSWD98.d.ts → index-p2TR_iWJ.d.ts} +5 -5
  47. package/dist/{index-NPeSWD98.d.ts.map → index-p2TR_iWJ.d.ts.map} +1 -1
  48. package/dist/index.d.ts +21 -21
  49. package/dist/index.js +8 -8
  50. package/dist/{llm-client--GR4JbZE.js → llm-client-BNcP4v08.js} +2 -2
  51. package/dist/{llm-client--GR4JbZE.js.map → llm-client-BNcP4v08.js.map} +1 -1
  52. package/dist/{llm-client-B_nIBlYo.d.ts → llm-client-BiK4HW0u.d.ts} +2 -2
  53. package/dist/{llm-client-B_nIBlYo.d.ts.map → llm-client-BiK4HW0u.d.ts.map} +1 -1
  54. package/dist/meta-eval/index.d.ts +2 -2
  55. package/dist/multishot/index.d.ts +1 -1
  56. package/dist/openapi.json +1 -1
  57. package/dist/{release-report-Crg9oFJ0.d.ts → release-report-DfmKSIEE.d.ts} +3 -3
  58. package/dist/{release-report-Crg9oFJ0.d.ts.map → release-report-DfmKSIEE.d.ts.map} +1 -1
  59. package/dist/{replay-RE97Ckjl.d.ts → replay-BI6CVKkp.d.ts} +2 -2
  60. package/dist/{replay-RE97Ckjl.d.ts.map → replay-BI6CVKkp.d.ts.map} +1 -1
  61. package/dist/reporting.d.ts +4 -4
  62. package/dist/{researcher-Q5rpPqZY.d.ts → researcher-DMimgHtN.d.ts} +4 -4
  63. package/dist/{researcher-Q5rpPqZY.d.ts.map → researcher-DMimgHtN.d.ts.map} +1 -1
  64. package/dist/{reward-hacking-CW-3HN0n.d.ts → reward-hacking-D-QqXvg-.d.ts} +2 -2
  65. package/dist/{reward-hacking-CW-3HN0n.d.ts.map → reward-hacking-D-QqXvg-.d.ts.map} +1 -1
  66. package/dist/rl.d.ts +5 -5
  67. package/dist/rl.js +1 -1
  68. package/dist/rollout/index.d.ts +1 -1
  69. package/dist/{rubric-predictive-validity-lXLmashy.d.ts → rubric-predictive-validity-C1dCLcvb.d.ts} +2 -2
  70. package/dist/{rubric-predictive-validity-lXLmashy.d.ts.map → rubric-predictive-validity-C1dCLcvb.d.ts.map} +1 -1
  71. package/dist/{run-evidence-oByzm-dE.d.ts → run-evidence-DokQtX0-.d.ts} +2 -2
  72. package/dist/{run-evidence-oByzm-dE.d.ts.map → run-evidence-DokQtX0-.d.ts.map} +1 -1
  73. package/dist/run-record-CN8Zd21B.js.map +1 -1
  74. package/dist/{run-record-BJnYdTxO.d.ts → run-record-DcObtIGh.d.ts} +4 -14
  75. package/dist/run-record-DcObtIGh.d.ts.map +1 -0
  76. package/dist/{runtime-trajectory-BXxG4lyi.d.ts → runtime-trajectory-BW9Wszb-.d.ts} +2 -2
  77. package/dist/{runtime-trajectory-BXxG4lyi.d.ts.map → runtime-trajectory-BW9Wszb-.d.ts.map} +1 -1
  78. package/dist/{semantic-concept-judge-b5m3irbR.js → semantic-concept-judge-DKCtoOz8.js} +4 -4
  79. package/dist/{semantic-concept-judge-b5m3irbR.js.map → semantic-concept-judge-DKCtoOz8.js.map} +1 -1
  80. package/dist/{server-m5D9cvnG.js → server-Dc_lsOYd.js} +3 -3
  81. package/dist/{server-m5D9cvnG.js.map → server-Dc_lsOYd.js.map} +1 -1
  82. package/dist/{skill-usage-D5mlWdAJ.d.ts → skill-usage-BaaxFSJR.d.ts} +5 -5
  83. package/dist/{skill-usage-D5mlWdAJ.d.ts.map → skill-usage-BaaxFSJR.d.ts.map} +1 -1
  84. package/dist/{skillopt-optimization-method-D0MVjJdP.d.ts → skillopt-optimization-method-C9M_lxdo.d.ts} +9 -9
  85. package/dist/{skillopt-optimization-method-D0MVjJdP.d.ts.map → skillopt-optimization-method-C9M_lxdo.d.ts.map} +1 -1
  86. package/dist/{skillopt-optimization-method-eOJL2570.js → skillopt-optimization-method-CQlz8GQM.js} +3 -3
  87. package/dist/{skillopt-optimization-method-eOJL2570.js.map → skillopt-optimization-method-CQlz8GQM.js.map} +1 -1
  88. package/dist/{statistics-Cmj6nynr.d.ts → statistics-DbvkkDPa.d.ts} +2 -2
  89. package/dist/{statistics-Cmj6nynr.d.ts.map → statistics-DbvkkDPa.d.ts.map} +1 -1
  90. package/dist/{summary-report-CWwB_LiV.d.ts → summary-report-DnUcjVpV.d.ts} +2 -2
  91. package/dist/{summary-report-CWwB_LiV.d.ts.map → summary-report-DnUcjVpV.d.ts.map} +1 -1
  92. package/dist/traces.d.ts +2 -2
  93. package/dist/{types-CsD5nTfV.d.ts → types-BokuXvOG.d.ts} +4 -4
  94. package/dist/{types-CsD5nTfV.d.ts.map → types-BokuXvOG.d.ts.map} +1 -1
  95. package/dist/{types-DGsxbAEd.d.ts → types-Cc3qbqzj.d.ts} +3 -3
  96. package/dist/{types-DGsxbAEd.d.ts.map → types-Cc3qbqzj.d.ts.map} +1 -1
  97. package/dist/wire/index.d.ts +2 -2
  98. package/dist/wire/index.js +1 -1
  99. package/package.json +2 -2
  100. package/dist/cost-ledger-DIgQUFZZ.js.map +0 -1
  101. package/dist/cost-ledger-Dye6jCgg.d.ts.map +0 -1
  102. package/dist/default-registry-C-vFCSEc.js.map +0 -1
  103. package/dist/run-record-BJnYdTxO.d.ts.map +0 -1
@@ -1,9 +1,10 @@
1
1
  import { s as ValidationError } from "../errors-8YnH8WlF.js";
2
- import { L as createChatClient, t as buildDefaultAnalystRegistry } from "../default-registry-C-vFCSEc.js";
2
+ import { L as createChatClient, t as buildDefaultAnalystRegistry } from "../default-registry-B1JcpnRv.js";
3
+ import { i as CostLedger } from "../cost-ledger-ZAa_P4r0.js";
3
4
  import { LLM_MODEL_ATTR_KEYS, SPAN_KIND_ATTR_KEYS } from "../trace-attributes.js";
4
5
  import { b as classifyOtlpSpanRole, x as isOtlpModelCall } from "../tools-BmuN627J.js";
5
6
  import { r as mapConcurrentRange } from "../concurrency-MUjT7VjM.js";
6
- import { B as surfaceContentHash, Ct as llmJudge, M as compareOptimizationMethods, O as composeGate, Q as inMemoryCampaignStorage, S as defaultProductionGate, T as heldoutSignificance, V as surfaceHash, X as createRunCostLedger, Y as runCampaign, Z as fsCampaignStorage, _ as buildEvidenceVector, a as emitLoopProvenance, b as powerPreflight, ct as campaignSplitDigest, d as runImprovementLoop, dt as REFERENCE_EQUIVALENCE_INPUT_LIMITS, ft as REFERENCE_EQUIVALENCE_JUDGE_VERSION, g as gepaOptimizationMethod, j as assertOptimizationResult, k as externalTextOptimizationMethod, mt as runReferenceEquivalenceJudge, o as loopProvenanceArgsFromResult, p as runEval, pt as createReferenceEquivalenceJudge, rt as resolveRunDir, t as skillOptOptimizationMethod, v as paretoPolicy, x as heldOutGate, y as paretoSignificanceGate } from "../skillopt-optimization-method-eOJL2570.js";
7
+ import { B as surfaceContentHash, Ct as llmJudge, M as compareOptimizationMethods, O as composeGate, Q as inMemoryCampaignStorage, S as defaultProductionGate, T as heldoutSignificance, V as surfaceHash, X as createRunCostLedger, Y as runCampaign, Z as fsCampaignStorage, _ as buildEvidenceVector, a as emitLoopProvenance, b as powerPreflight, ct as campaignSplitDigest, d as runImprovementLoop, dt as REFERENCE_EQUIVALENCE_INPUT_LIMITS, ft as REFERENCE_EQUIVALENCE_JUDGE_VERSION, g as gepaOptimizationMethod, j as assertOptimizationResult, k as externalTextOptimizationMethod, mt as runReferenceEquivalenceJudge, o as loopProvenanceArgsFromResult, p as runEval, pt as createReferenceEquivalenceJudge, rt as resolveRunDir, t as skillOptOptimizationMethod, v as paretoPolicy, x as heldOutGate, y as paretoSignificanceGate } from "../skillopt-optimization-method-CQlz8GQM.js";
7
8
  import { v as pairedBootstrap } from "../statistics-CnnxdpOg.js";
8
9
  import { i as parseRunRecordSafe, r as modelHasSnapshot } from "../run-record-CN8Zd21B.js";
9
10
  import { i as summarizeTraceErrors, n as recordAggregateMeasurements, r as summarizeExecutionMeasurements, t as readTaskFailureLabels } from "../task-failure-attributes-CQZlB3et.js";
@@ -14,7 +15,7 @@ import { a as fromPiSession, c as observeCodeAgentSession, i as fromOpenCodeSess
14
15
  import { t as createHostedClient } from "../client-CYzbdJOZ.js";
15
16
  import { dirname, join } from "node:path";
16
17
  import { mkdir, readFile, readdir, stat, writeFile } from "node:fs/promises";
17
- import { agentCandidateBenchmarkSuiteSchema, agentCandidateBenchmarkTaskSchema, agentCandidateBundleSchema, agentCandidateEvaluationPolicySchema, agentCandidateExperimentSchema, agentImprovementMeasuredComparisonSchema, agentProfileImprovementExperimentSchema, agentProfileImprovementMeasuredComparisonSchema, agentProfileImprovementRunCellSchema, agentProfileImprovementRunReceiptSchema, agentProfileImprovementSuiteInputsSchema, agentProfileImprovementSuiteSchema, agentProfileImprovementTaskSchema, candidateExecutionEvidenceSchema, canonicalCandidateDigest, canonicalCandidateJson, omitTopLevelDigest } from "@tangle-network/agent-interface";
18
+ import { agentCandidateBenchmarkSuiteSchema, agentCandidateBenchmarkTaskSchema, agentCandidateBundleSchema, agentCandidateEvaluationPolicySchema, agentCandidateExperimentSchema, agentImprovementMeasuredComparisonSchema, agentProfileImprovementExperimentSchema, agentProfileImprovementMeasuredComparisonSchema, agentProfileImprovementRunCellSchema, agentProfileImprovementRunReceiptSchema, agentProfileImprovementSuiteInputsSchema, agentProfileImprovementSuiteSchema, agentProfileImprovementTaskSchema, candidateExecutionEvidenceSchema, canonicalCandidateDigest, canonicalCandidateJson, numbersApproximatelyEqual, omitTopLevelDigest } from "@tangle-network/agent-interface";
18
19
  //#region src/contract/self-improve.ts
19
20
  /**
20
21
  * Run one complete improvement job.
@@ -546,6 +547,19 @@ function requirePositiveInteger(value, field) {
546
547
  return value;
547
548
  }
548
549
  //#endregion
550
+ //#region src/contract/fixed-spend.ts
551
+ function addFixedSpend(left, right) {
552
+ return {
553
+ inputTokens: left.inputTokens + right.inputTokens,
554
+ outputTokens: left.outputTokens + right.outputTokens,
555
+ cachedInputTokens: left.cachedInputTokens + right.cachedInputTokens,
556
+ reasoningTokens: left.reasoningTokens + right.reasoningTokens,
557
+ modelCalls: left.modelCalls + right.modelCalls,
558
+ costUsdNanos: left.costUsdNanos + right.costUsdNanos,
559
+ costProvenance: left.costProvenance === "observed" && right.costProvenance === "observed" ? "observed" : "estimated"
560
+ };
561
+ }
562
+ //#endregion
549
563
  //#region src/contract/concurrent-map.ts
550
564
  /** Map both arms while bounding the actual executions, not the number of pairs. */
551
565
  async function mapPairedConcurrent(options) {
@@ -568,6 +582,62 @@ async function mapPairedConcurrent(options) {
568
582
  }));
569
583
  }
570
584
  //#endregion
585
+ //#region src/contract/paid-paired-measurement.ts
586
+ /**
587
+ * Reserve the complete signed suite before its first cell starts.
588
+ *
589
+ * The host executor receives each task's signed cap through `execute`; this
590
+ * batch reservation prevents a complete comparison from turning into a partial
591
+ * one after earlier analysis or search has used the remaining budget.
592
+ */
593
+ async function runPaidPairedMeasurement(options) {
594
+ assertNonNegativeFinite(options.maximumCostUsd, "maximumCostUsd");
595
+ const costLedger = resolveCostLedger(options);
596
+ const startedAt = performance.now();
597
+ const paid = await costLedger.runPaidCall({
598
+ callId: options.call.callId,
599
+ channel: options.call.channel,
600
+ phase: options.call.phase,
601
+ actor: options.call.actor,
602
+ model: options.call.model,
603
+ maximumCharge: { externallyEnforcedMaximumUsd: options.maximumCostUsd },
604
+ ...options.call.tags ? { tags: options.call.tags } : {},
605
+ ...options.signal ? { signal: options.signal } : {},
606
+ execute: async (signal) => await mapPairedConcurrent({
607
+ count: options.count,
608
+ maxConcurrency: options.maxConcurrency,
609
+ label: options.label,
610
+ signal,
611
+ map: options.execute
612
+ }),
613
+ receipt: options.receipt
614
+ });
615
+ const wallDurationMs = performance.now() - startedAt;
616
+ if (!paid.succeeded) throw paid.error;
617
+ if (paid.receipt.costUnknown) throw new Error(`${options.label} did not capture the complete measurement cost`);
618
+ return {
619
+ measurements: paid.value,
620
+ wallDurationMs,
621
+ cost: paid.receipt.actualCostUsd === void 0 ? {
622
+ kind: "estimated",
623
+ usd: paid.receipt.costUsd
624
+ } : {
625
+ kind: "observed",
626
+ usd: paid.receipt.costUsd
627
+ }
628
+ };
629
+ }
630
+ function resolveCostLedger(options) {
631
+ if (options.budgetUsd === void 0) return options.costLedger ?? new CostLedger();
632
+ assertNonNegativeFinite(options.budgetUsd, "budgetUsd");
633
+ if (!options.costLedger) throw new Error(`${options.label} with a policy budget requires one shared CostLedger`);
634
+ if (options.costLedger.costCeilingUsd !== options.budgetUsd) throw new Error(`${options.label} CostLedger ceiling must equal the frozen policy budget`);
635
+ return options.costLedger;
636
+ }
637
+ function assertNonNegativeFinite(value, label) {
638
+ if (!Number.isFinite(value) || value < 0) throw new Error(`${label} must be a non-negative finite number`);
639
+ }
640
+ //#endregion
571
641
  //#region src/contract/measured-comparison.ts
572
642
  /** Content-address one task before any measured execution can see it. */
573
643
  function sealCandidateBenchmarkTask(material) {
@@ -614,12 +684,23 @@ function verifyCandidateExperiment(input) {
614
684
  async function runCandidateExperiment(options) {
615
685
  const experiment = verifyCandidateExperiment(options.experiment);
616
686
  const { suite, tasks } = experiment.benchmark;
617
- return (await mapPairedConcurrent({
687
+ const run = await runPaidPairedMeasurement({
618
688
  count: suite.taskDigests.length * suite.reps,
619
689
  maxConcurrency: options.maxConcurrency ?? 2,
620
690
  label: "candidate experiment",
691
+ budgetUsd: experiment.policy.budgetUsd,
692
+ ...options.costLedger ? { costLedger: options.costLedger } : {},
693
+ maximumCostUsd: candidateMeasurementMaximumCostUsd(experiment),
694
+ call: {
695
+ callId: `candidate-measurement:${experiment.digest}`,
696
+ channel: "measurement",
697
+ phase: "heldout",
698
+ actor: experiment.digest,
699
+ model: candidateMeasurementModel(experiment),
700
+ tags: { experimentDigest: experiment.digest }
701
+ },
621
702
  ...options.signal ? { signal: options.signal } : {},
622
- async map(index, arm, signal) {
703
+ async execute(index, arm, signal) {
623
704
  const taskIndex = Math.floor(index / suite.reps);
624
705
  const repetition = index % suite.reps;
625
706
  const task = tasks[taskIndex];
@@ -630,7 +711,7 @@ async function runCandidateExperiment(options) {
630
711
  taskIndex,
631
712
  repetition
632
713
  };
633
- return options.execute({
714
+ return verifyExecutionEvidence(await options.execute({
634
715
  experiment,
635
716
  arm,
636
717
  bundle: experiment[arm],
@@ -638,9 +719,22 @@ async function runCandidateExperiment(options) {
638
719
  benchmarkCell,
639
720
  seed,
640
721
  signal
641
- });
722
+ }));
723
+ },
724
+ receipt(measurements) {
725
+ return candidateMeasurementCostReceipt(measurements, candidateMeasurementModel(experiment));
726
+ }
727
+ });
728
+ return {
729
+ measurements: run.measurements.map((measurement, index) => verifyMeasurement(experiment, measurement, index)),
730
+ measurement: {
731
+ wallDurationMs: run.wallDurationMs,
732
+ cost: {
733
+ usd: run.cost.usd,
734
+ provenance: run.cost.kind
735
+ }
642
736
  }
643
- })).map((measurement, index) => verifyMeasurement(experiment, measurement, index));
737
+ };
644
738
  }
645
739
  /**
646
740
  * Calculate the shared paired decision from any complete receipt shape.
@@ -651,8 +745,11 @@ async function runCandidateExperiment(options) {
651
745
  */
652
746
  function evaluatePairedMeasurements(options) {
653
747
  if (options.measurements.length === 0) throw new Error("paired measurement evaluation requires at least one paired cell");
654
- const additionalCostUsd = options.additionalCostUsd ?? 0;
655
- if (!Number.isFinite(additionalCostUsd) || additionalCostUsd < 0) throw new Error("paired measurement evaluation additionalCostUsd must be a non-negative number");
748
+ const preparationCost = options.preparationCost ?? {
749
+ usd: 0,
750
+ provenance: "observed"
751
+ };
752
+ if (!Number.isFinite(preparationCost.usd) || preparationCost.usd < 0) throw new Error("paired measurement evaluation preparation cost must be a non-negative number");
656
753
  if (typeof options.sharedScorerChannel !== "boolean") throw new Error("paired measurement evaluation sharedScorerChannel must be a boolean");
657
754
  const policy = agentCandidateEvaluationPolicySchema.parse(options.policy);
658
755
  const measurements = options.measurements.map((measurement, index) => projectPairedMeasurement(measurement, index, options.adapter));
@@ -738,12 +835,22 @@ function evaluatePairedMeasurements(options) {
738
835
  const guardedDimensions = new Set(criticalDimensions);
739
836
  const missingCriticalDimensions = criticalDimensions.filter((dimension) => !dimensions.includes(dimension));
740
837
  const regressions = objectives.filter((objective) => objective.kind === "dimension" && guardedDimensions.has(objective.name) && objective.availability === "measured" && objective.confidenceInterval.lower < -regressionTolerance);
741
- const executionCostUsd = measurements.reduce((sum, measurement) => sum + measurement.baseline.costUsd + measurement.candidate.costUsd, 0);
742
- const executionDurationMs = measurements.reduce((sum, measurement) => sum + measurement.baseline.latencyMs + measurement.candidate.latencyMs, 0);
838
+ const measurementCostUsd = measurements.reduce((sum, measurement) => sum + measurement.baseline.costUsd + measurement.candidate.costUsd, 0);
839
+ const measurementWorkDurationMs = measurements.reduce((sum, measurement) => sum + measurement.baseline.latencyMs + measurement.candidate.latencyMs, 0);
743
840
  const incompleteRuns = measurements.flatMap((measurement) => [measurement.baseline, measurement.candidate]).filter((run) => !run.completed);
744
841
  const failedCandidateResults = measurements.filter((measurement) => !measurement.candidate.passed);
745
- const totalCostUsd = executionCostUsd + additionalCostUsd;
746
- const budgetPassed = budgetUsd === void 0 || totalCostUsd <= budgetUsd;
842
+ const derivedMeasurementCost = {
843
+ usd: measurementCostUsd,
844
+ provenance: measurements.every((measurement) => measurement.baseline.costProvenance === "observed" && measurement.candidate.costProvenance === "observed") ? "observed" : "estimated"
845
+ };
846
+ const measurementCost = options.measurementCost ?? derivedMeasurementCost;
847
+ assertKnownCost(measurementCost, "paired measurement cost");
848
+ if (options.measurementCost !== void 0 && (measurementCost.provenance !== derivedMeasurementCost.provenance || !numbersApproximatelyEqual(measurementCost.usd, derivedMeasurementCost.usd))) throw new Error("paired measurement cost does not match its signed receipts");
849
+ const totalCost = {
850
+ usd: preparationCost.usd + measurementCost.usd,
851
+ provenance: preparationCost.provenance === "observed" && measurementCost.provenance === "observed" ? "observed" : "estimated"
852
+ };
853
+ const budgetPassed = budgetUsd === void 0 || totalCost.usd < budgetUsd || numbersApproximatelyEqual(totalCost.usd, budgetUsd);
747
854
  const checks = [
748
855
  {
749
856
  name: "paired-significance",
@@ -778,7 +885,7 @@ function evaluatePairedMeasurements(options) {
778
885
  ...missingCriticalDimensions.length === 0 ? [] : [`critical dimensions missing: ${missingCriticalDimensions.join(", ")}`],
779
886
  ...incompleteRuns.length === 0 ? [] : [`${incompleteRuns.length} benchmark executions did not exit successfully`],
780
887
  ...failedCandidateResults.length === 0 ? [] : [`candidate failed ${failedCandidateResults.length} benchmark tasks`],
781
- ...budgetPassed ? [] : [`total cost ${totalCostUsd} exceeded budget ${budgetUsd}`]
888
+ ...budgetPassed ? [] : [`total cost ${totalCost.usd} exceeded budget ${budgetUsd}`]
782
889
  ];
783
890
  return {
784
891
  overall: {
@@ -802,9 +909,9 @@ function evaluatePairedMeasurements(options) {
802
909
  sharedScorerChannel: options.sharedScorerChannel,
803
910
  reason: power.reason
804
911
  },
805
- executionCostUsd,
806
- totalCostUsd,
807
- executionDurationMs
912
+ measurementCost,
913
+ totalCost,
914
+ measurementWorkDurationMs
808
915
  };
809
916
  }
810
917
  /** Build the only publishable comparison: paired statistics over Runtime receipts. */
@@ -815,7 +922,8 @@ function measuredComparisonFromCandidateExperiment(options) {
815
922
  if (measurements.length !== expectedN) throw new Error(`candidate experiment is incomplete (${measurements.length}/${expectedN} paired cells)`);
816
923
  verifyStableProfileMaterialization(measurements);
817
924
  if (!options.runId.trim()) throw new Error("candidate experiment runId is required");
818
- const searchCostUsd = options.searchCostUsd ?? 0;
925
+ assertPhaseAccounting(options.preparation, "candidate preparation");
926
+ assertPhaseAccounting(options.measurement, "candidate measurement");
819
927
  const evaluation = evaluatePairedMeasurements({
820
928
  measurements: measurements.map((measurement, index) => ({
821
929
  cellId: cellIds(experiment)[index],
@@ -824,12 +932,10 @@ function measuredComparisonFromCandidateExperiment(options) {
824
932
  policy: experiment.policy,
825
933
  adapter: candidateExecutionEvidenceAdapter,
826
934
  sharedScorerChannel: true,
827
- additionalCostUsd: searchCostUsd
935
+ preparationCost: options.preparation.cost,
936
+ measurementCost: options.measurement.cost
828
937
  });
829
938
  const diff = deriveCandidateBundleDiff(experiment);
830
- const searchDurationMs = options.searchDurationMs ?? 0;
831
- const totalCostUsd = evaluation.totalCostUsd;
832
- const durationMs = evaluation.executionDurationMs + searchDurationMs;
833
939
  const provisional = agentImprovementMeasuredComparisonSchema.parse({
834
940
  kind: "agent-improvement-measured-comparison",
835
941
  experiment,
@@ -850,12 +956,16 @@ function measuredComparisonFromCandidateExperiment(options) {
850
956
  diff,
851
957
  evaluation: {
852
958
  generationsExplored: options.generationsExplored ?? 0,
853
- searchDurationMs,
854
- executionDurationMs: evaluation.executionDurationMs,
855
- durationMs,
856
- searchCostUsd,
857
- executionCostUsd: evaluation.executionCostUsd,
858
- totalCostUsd
959
+ preparation: options.preparation,
960
+ measurement: {
961
+ wallDurationMs: options.measurement.wallDurationMs,
962
+ workDurationMs: evaluation.measurementWorkDurationMs,
963
+ cost: evaluation.measurementCost
964
+ },
965
+ total: {
966
+ wallDurationMs: options.preparation.wallDurationMs + options.measurement.wallDurationMs,
967
+ cost: evaluation.totalCost
968
+ }
859
969
  },
860
970
  ...options.metadata ? { metadata: options.metadata } : {}
861
971
  });
@@ -880,12 +990,52 @@ function verifyCandidateExperimentComparison(input) {
880
990
  runId: comparison.provenance.runId,
881
991
  ...comparison.candidate ? { candidate: comparison.candidate } : {},
882
992
  generationsExplored: comparison.evaluation.generationsExplored,
883
- searchDurationMs: comparison.evaluation.searchDurationMs,
884
- searchCostUsd: comparison.evaluation.searchCostUsd,
993
+ preparation: comparison.evaluation.preparation,
994
+ measurement: {
995
+ wallDurationMs: comparison.evaluation.measurement.wallDurationMs,
996
+ cost: comparison.evaluation.measurement.cost
997
+ },
885
998
  ...comparison.metadata ? { metadata: comparison.metadata } : {}
886
999
  })) !== canonicalCandidateDigest(comparison)) throw new Error("candidate experiment comparison does not match its Runtime receipts");
887
1000
  return comparison;
888
1001
  }
1002
+ function candidateMeasurementMaximumCostUsd(experiment) {
1003
+ const maximum = experiment.benchmark.tasks.reduce((sum, task) => sum + task.limits.maxCostUsd * experiment.benchmark.suite.reps * 2, 0);
1004
+ if (!Number.isFinite(maximum) || maximum < 0) throw new Error("candidate measurement maximum cost is invalid");
1005
+ return maximum;
1006
+ }
1007
+ function candidateMeasurementModel(experiment) {
1008
+ const models = [...new Set(experiment.benchmark.tasks.map((task) => task.model.model))];
1009
+ return models.length === 1 ? models[0] : "multiple-models";
1010
+ }
1011
+ function candidateMeasurementCostReceipt(measurements, model) {
1012
+ const usage = measurements.reduce((sum, measurement) => addFixedSpend(addFixedSpend(sum, combinedUsage(measurement.baseline)), combinedUsage(measurement.candidate)), {
1013
+ inputTokens: 0,
1014
+ outputTokens: 0,
1015
+ cachedInputTokens: 0,
1016
+ reasoningTokens: 0,
1017
+ modelCalls: 0,
1018
+ costUsdNanos: 0,
1019
+ costProvenance: "observed"
1020
+ });
1021
+ const costUsd = usage.costUsdNanos / 1e9;
1022
+ return {
1023
+ model,
1024
+ inputTokens: usage.inputTokens,
1025
+ outputTokens: usage.outputTokens,
1026
+ cachedTokens: usage.cachedInputTokens,
1027
+ reasoningTokens: usage.reasoningTokens,
1028
+ ...usage.costProvenance === "observed" ? { actualCostUsd: costUsd } : { estimatedCostUsd: costUsd }
1029
+ };
1030
+ }
1031
+ function assertPhaseAccounting(accounting, label) {
1032
+ if (!Number.isFinite(accounting.wallDurationMs) || accounting.wallDurationMs < 0) throw new Error(`${label} wall duration must be a non-negative number`);
1033
+ assertKnownCost(accounting.cost, `${label} cost`);
1034
+ }
1035
+ function assertKnownCost(cost, label) {
1036
+ if (!Number.isFinite(cost.usd) || cost.usd < 0) throw new Error(`${label} must be a non-negative number`);
1037
+ if (cost.provenance !== "observed" && cost.provenance !== "estimated") throw new Error(`${label} provenance must be observed or estimated`);
1038
+ }
889
1039
  function deriveCandidateBundleDiff(experiment) {
890
1040
  const changed = [
891
1041
  "profile",
@@ -1061,10 +1211,13 @@ function projectRun(run, adapter, label) {
1061
1211
  const completed = adapter.completed(run);
1062
1212
  const passed = adapter.passed(run);
1063
1213
  if (typeof completed !== "boolean" || typeof passed !== "boolean") throw new Error(`${label} completion and pass values must be booleans`);
1214
+ const costProvenance = adapter.costProvenance(run);
1215
+ if (costProvenance !== "observed" && costProvenance !== "estimated") throw new Error(`${label} cost provenance must be observed or estimated`);
1064
1216
  return {
1065
1217
  score: finiteMeasurement(adapter.score(run), `${label} score`),
1066
1218
  dimensions,
1067
1219
  costUsd: nonNegativeMeasurement(adapter.costUsd(run), `${label} cost`),
1220
+ costProvenance,
1068
1221
  latencyMs: nonNegativeMeasurement(adapter.latencyMs(run), `${label} latency`),
1069
1222
  completed,
1070
1223
  passed
@@ -1149,6 +1302,7 @@ const candidateExecutionEvidenceAdapter = {
1149
1302
  score: (evidence) => evidence.receipt.benchmarkResult.material.score,
1150
1303
  dimensions: (evidence) => evidence.receipt.benchmarkResult.material.dimensions,
1151
1304
  costUsd: costFromEvidence,
1305
+ costProvenance: (evidence) => combinedUsage(evidence).costProvenance,
1152
1306
  latencyMs: latencyFromEvidence,
1153
1307
  completed: completedSuccessfully,
1154
1308
  passed: (evidence) => evidence.receipt.benchmarkResult.material.passed
@@ -1162,14 +1316,7 @@ function latencyFromEvidence(evidence) {
1162
1316
  function combinedUsage(evidence) {
1163
1317
  const candidate = evidence.receipt.modelSettlement.material.usage;
1164
1318
  const grader = evidence.receipt.benchmarkResult.material.grading.usage;
1165
- return {
1166
- inputTokens: candidate.inputTokens + grader.inputTokens,
1167
- outputTokens: candidate.outputTokens + grader.outputTokens,
1168
- cachedInputTokens: candidate.cachedInputTokens + grader.cachedInputTokens,
1169
- reasoningTokens: candidate.reasoningTokens + grader.reasoningTokens,
1170
- modelCalls: candidate.modelCalls + grader.modelCalls,
1171
- costUsdNanos: candidate.costUsdNanos + grader.costUsdNanos
1172
- };
1319
+ return addFixedSpend(candidate, grader);
1173
1320
  }
1174
1321
  function cellIds(experiment) {
1175
1322
  const { suite, tasks } = experiment.benchmark;
@@ -1235,15 +1382,45 @@ function verifyAgentProfileImprovementExperiment(input) {
1235
1382
  */
1236
1383
  async function runAgentProfileImprovementExperiment(options) {
1237
1384
  const experiment = verifyAgentProfileImprovementExperiment(options.experiment);
1238
- return (await mapPairedConcurrent({
1385
+ const run = await runPaidPairedMeasurement({
1239
1386
  count: experiment.benchmark.suite.taskDigests.length * experiment.benchmark.suite.reps,
1240
1387
  maxConcurrency: options.maxConcurrency ?? 2,
1241
1388
  label: "profile improvement experiment",
1389
+ budgetUsd: experiment.policy.budgetUsd,
1390
+ ...options.costLedger ? { costLedger: options.costLedger } : {},
1391
+ maximumCostUsd: profileMeasurementMaximumCostUsd(experiment),
1392
+ call: {
1393
+ callId: `profile-improvement-measurement:${experiment.digest}`,
1394
+ channel: "measurement",
1395
+ phase: "heldout",
1396
+ actor: experiment.executionRef.identity,
1397
+ model: profileMeasurementModel(experiment),
1398
+ tags: {
1399
+ experimentDigest: experiment.digest,
1400
+ executorDigest: experiment.executionRef.digest
1401
+ }
1402
+ },
1242
1403
  ...options.signal ? { signal: options.signal } : {},
1243
- map(index, arm, signal) {
1244
- return options.execute(profileExecutionInput(experiment, arm, index, signal));
1404
+ async execute(index, arm, signal) {
1405
+ const expected = profileExecutionInput(experiment, arm, index, signal);
1406
+ const receipt = agentProfileImprovementRunReceiptSchema.parse(await options.execute(expected));
1407
+ verifyProfileReceiptContract(receipt, expected, index);
1408
+ return receipt;
1409
+ },
1410
+ receipt(measurements) {
1411
+ return profileMeasurementCostReceipt(measurements, profileMeasurementModel(experiment));
1245
1412
  }
1246
- })).map((measurement, index) => verifyProfileMeasurement(experiment, measurement, index));
1413
+ });
1414
+ return {
1415
+ measurements: run.measurements.map((measurement, index) => verifyProfileMeasurement(experiment, measurement, index)),
1416
+ measurement: {
1417
+ wallDurationMs: run.wallDurationMs,
1418
+ cost: {
1419
+ usd: run.cost.usd,
1420
+ provenance: run.cost.kind
1421
+ }
1422
+ }
1423
+ };
1247
1424
  }
1248
1425
  /** Build the only publishable profile comparison from complete host receipts. */
1249
1426
  function measuredComparisonFromAgentProfileImprovementExperiment(options) {
@@ -1252,8 +1429,6 @@ function measuredComparisonFromAgentProfileImprovementExperiment(options) {
1252
1429
  const expectedCount = experiment.benchmark.suite.taskDigests.length * experiment.benchmark.suite.reps;
1253
1430
  if (measurements.length !== expectedCount) throw new Error(`profile improvement experiment is incomplete (${measurements.length}/${expectedCount} paired cells)`);
1254
1431
  if (!options.runId.trim()) throw new Error("profile improvement experiment runId is required");
1255
- const searchCostUsd = nonNegative(options.searchCostUsd ?? 0, "searchCostUsd");
1256
- const searchDurationMs = nonNegative(options.searchDurationMs ?? 0, "searchDurationMs");
1257
1432
  const evaluation = evaluatePairedMeasurements({
1258
1433
  measurements: measurements.map((measurement, index) => ({
1259
1434
  cellId: profileCellId(experiment, index),
@@ -1262,7 +1437,8 @@ function measuredComparisonFromAgentProfileImprovementExperiment(options) {
1262
1437
  policy: experiment.policy,
1263
1438
  adapter: profileReceiptAdapter,
1264
1439
  sharedScorerChannel: true,
1265
- additionalCostUsd: searchCostUsd
1440
+ preparationCost: options.preparation.cost,
1441
+ measurementCost: options.measurement.cost
1266
1442
  });
1267
1443
  const provisional = agentProfileImprovementMeasuredComparisonSchema.parse({
1268
1444
  kind: "agent-profile-improvement-measured-comparison",
@@ -1284,12 +1460,16 @@ function measuredComparisonFromAgentProfileImprovementExperiment(options) {
1284
1460
  diff: canonicalCandidateJson(experiment.change),
1285
1461
  evaluation: {
1286
1462
  generationsExplored: options.generationsExplored ?? 0,
1287
- searchDurationMs,
1288
- executionDurationMs: evaluation.executionDurationMs,
1289
- durationMs: evaluation.executionDurationMs + searchDurationMs,
1290
- searchCostUsd,
1291
- executionCostUsd: evaluation.executionCostUsd,
1292
- totalCostUsd: evaluation.totalCostUsd
1463
+ preparation: options.preparation,
1464
+ measurement: {
1465
+ wallDurationMs: options.measurement.wallDurationMs,
1466
+ workDurationMs: evaluation.measurementWorkDurationMs,
1467
+ cost: evaluation.measurementCost
1468
+ },
1469
+ total: {
1470
+ wallDurationMs: options.preparation.wallDurationMs + options.measurement.wallDurationMs,
1471
+ cost: evaluation.totalCost
1472
+ }
1293
1473
  },
1294
1474
  ...options.metadata ? { metadata: options.metadata } : {}
1295
1475
  });
@@ -1314,12 +1494,48 @@ function verifyAgentProfileImprovementExperimentComparison(input) {
1314
1494
  runId: comparison.provenance.runId,
1315
1495
  ...comparison.candidate ? { candidate: comparison.candidate } : {},
1316
1496
  generationsExplored: comparison.evaluation.generationsExplored,
1317
- searchDurationMs: comparison.evaluation.searchDurationMs,
1318
- searchCostUsd: comparison.evaluation.searchCostUsd,
1497
+ preparation: comparison.evaluation.preparation,
1498
+ measurement: {
1499
+ wallDurationMs: comparison.evaluation.measurement.wallDurationMs,
1500
+ cost: comparison.evaluation.measurement.cost
1501
+ },
1319
1502
  ...comparison.metadata ? { metadata: comparison.metadata } : {}
1320
1503
  })) !== canonicalCandidateDigest(comparison)) throw new Error("profile improvement comparison does not match its Runtime receipts");
1321
1504
  return comparison;
1322
1505
  }
1506
+ function profileMeasurementMaximumCostUsd(experiment) {
1507
+ const pairs = experiment.benchmark.suite.reps;
1508
+ const maximum = experiment.benchmark.tasks.reduce((sum, task) => sum + task.limits.maxCostUsd * pairs * 2, 0);
1509
+ if (!Number.isFinite(maximum) || maximum < 0) throw new Error("profile improvement measurement maximum cost is invalid");
1510
+ return maximum;
1511
+ }
1512
+ function profileMeasurementModel(experiment) {
1513
+ const models = [...new Set(experiment.benchmark.tasks.map((task) => task.model.model))];
1514
+ return models.length === 1 ? models[0] : "multiple-models";
1515
+ }
1516
+ function profileMeasurementCostReceipt(measurements, model) {
1517
+ const usage = measurements.reduce((sum, measurement) => addFixedSpend(addFixedSpend(sum, combinedProfileUsage(measurement.baseline)), combinedProfileUsage(measurement.candidate)), {
1518
+ inputTokens: 0,
1519
+ outputTokens: 0,
1520
+ cachedInputTokens: 0,
1521
+ reasoningTokens: 0,
1522
+ modelCalls: 0,
1523
+ costUsdNanos: 0,
1524
+ costProvenance: "observed"
1525
+ });
1526
+ const costUsd = usage.costUsdNanos / 1e9;
1527
+ return {
1528
+ model,
1529
+ inputTokens: usage.inputTokens,
1530
+ outputTokens: usage.outputTokens,
1531
+ cachedTokens: usage.cachedInputTokens,
1532
+ reasoningTokens: usage.reasoningTokens,
1533
+ ...usage.costProvenance === "observed" ? { actualCostUsd: costUsd } : { estimatedCostUsd: costUsd }
1534
+ };
1535
+ }
1536
+ function combinedProfileUsage(receipt) {
1537
+ return addFixedSpend(receipt.usage, receipt.grading.usage);
1538
+ }
1323
1539
  function profileExecutionInput(experiment, arm, index, signal) {
1324
1540
  const { task, taskIndex, repetition, seed } = profileCell(experiment, index);
1325
1541
  const stateDigest = experiment[arm].stateDigest;
@@ -1356,17 +1572,18 @@ function verifyProfileMeasurement(experiment, input, index) {
1356
1572
  const baseline = agentProfileImprovementRunReceiptSchema.parse(material.baseline);
1357
1573
  const candidate = agentProfileImprovementRunReceiptSchema.parse(material.candidate);
1358
1574
  if (baseline.runCell.digest !== expectedBaseline.runCell.digest || candidate.runCell.digest !== expectedCandidate.runCell.digest) throw new Error(`profile improvement measurement ${index} substituted a measured arm`);
1359
- verifyProfileReceiptTaskContract(baseline, expectedBaseline, index);
1360
- verifyProfileReceiptTaskContract(candidate, expectedCandidate, index);
1575
+ verifyProfileReceiptContract(baseline, expectedBaseline, index);
1576
+ verifyProfileReceiptContract(candidate, expectedCandidate, index);
1361
1577
  if (baseline.executionId === candidate.executionId || baseline.digest === candidate.digest) throw new Error(`profile improvement measurement ${index} reused one execution across arms`);
1362
1578
  return {
1363
1579
  baseline,
1364
1580
  candidate
1365
1581
  };
1366
1582
  }
1367
- function verifyProfileReceiptTaskContract(receipt, expected, index) {
1583
+ function verifyProfileReceiptContract(receipt, expected, index) {
1368
1584
  const task = expected.task;
1369
1585
  if (canonicalCandidateDigest(receipt.resolvedModel) !== canonicalCandidateDigest(task.model) || canonicalCandidateDigest(receipt.limits) !== canonicalCandidateDigest(task.limits) || canonicalCandidateDigest(receipt.grading.grader) !== canonicalCandidateDigest(task.grader)) throw new Error(`profile improvement measurement ${index} substituted its ${expected.arm} task contract`);
1586
+ if (canonicalCandidateDigest(receipt.executionRef) !== canonicalCandidateDigest(expected.experiment.executionRef)) throw new Error(`profile improvement measurement ${index} substituted its ${expected.arm} executor`);
1370
1587
  }
1371
1588
  function profileCell(experiment, index) {
1372
1589
  const { suite, tasks } = experiment.benchmark;
@@ -1389,14 +1606,11 @@ const profileReceiptAdapter = {
1389
1606
  score: (receipt) => receipt.grading.score,
1390
1607
  dimensions: (receipt) => receipt.grading.dimensions,
1391
1608
  costUsd: (receipt) => (receipt.usage.costUsdNanos + receipt.grading.usage.costUsdNanos) / 1e9,
1609
+ costProvenance: (receipt) => receipt.usage.costProvenance === "observed" && receipt.grading.usage.costProvenance === "observed" ? "observed" : "estimated",
1392
1610
  latencyMs: (receipt) => receipt.timing.durationMs + receipt.grading.timing.durationMs,
1393
1611
  completed: (receipt) => receipt.outcome.status === "succeeded",
1394
1612
  passed: (receipt) => receipt.grading.passed
1395
1613
  };
1396
- function nonNegative(value, label) {
1397
- if (!Number.isFinite(value) || value < 0) throw new Error(`${label} must be a non-negative number`);
1398
- return value;
1399
- }
1400
1614
  //#endregion
1401
1615
  //#region src/contract/intake/run-record-dir.ts
1402
1616
  /**