@tangle-network/agent-eval 0.131.1 → 0.132.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +7 -0
- package/dist/analyst/index.d.ts +5 -5
- package/dist/analyst/index.js +3 -3
- package/dist/{analyze-runs-CBnCgfse.d.ts → analyze-runs-AFDI5RI0.d.ts} +5 -5
- package/dist/{analyze-runs-CBnCgfse.d.ts.map → analyze-runs-AFDI5RI0.d.ts.map} +1 -1
- package/dist/belief-state/index.d.ts +4 -4
- package/dist/benchmarks/index.d.ts +1 -1
- package/dist/benchmarks/index.js +1 -1
- package/dist/{benchmarks-BDtRiOvE.js → benchmarks-DTrT3UH-.js} +3 -3
- package/dist/{benchmarks-BDtRiOvE.js.map → benchmarks-DTrT3UH-.js.map} +1 -1
- package/dist/campaign/index.d.ts +4 -4
- package/dist/campaign/index.js +2 -2
- package/dist/{campaign-CLG6y9jF.js → campaign-Cx6CfMR4.js} +4 -4
- package/dist/{campaign-CLG6y9jF.js.map → campaign-Cx6CfMR4.js.map} +1 -1
- package/dist/cli.js +1 -1
- package/dist/{client-w90OcvvR.d.ts → client-aZDHJiKO.d.ts} +4 -4
- package/dist/{client-w90OcvvR.d.ts.map → client-aZDHJiKO.d.ts.map} +1 -1
- package/dist/{code-agent-session-aYa3SGKz.d.ts → code-agent-session-D5URqc3_.d.ts} +2 -2
- package/dist/{code-agent-session-aYa3SGKz.d.ts.map → code-agent-session-D5URqc3_.d.ts.map} +1 -1
- package/dist/contract/index.d.ts +49 -22
- package/dist/contract/index.d.ts.map +1 -1
- package/dist/contract/index.js +275 -61
- package/dist/contract/index.js.map +1 -1
- package/dist/control.d.ts +1 -1
- package/dist/{cost-ledger-DIgQUFZZ.js → cost-ledger-ZAa_P4r0.js} +41 -7
- package/dist/cost-ledger-ZAa_P4r0.js.map +1 -0
- package/dist/{cost-ledger-Dye6jCgg.d.ts → cost-ledger-fGS_u_O1.d.ts} +18 -2
- package/dist/cost-ledger-fGS_u_O1.d.ts.map +1 -0
- package/dist/{default-registry-C-vFCSEc.js → default-registry-B1JcpnRv.js} +4 -13
- package/dist/default-registry-B1JcpnRv.js.map +1 -0
- package/dist/{default-registry-Cj1oUpLN.d.ts → default-registry-Cl3pHo4n.d.ts} +4 -4
- package/dist/{default-registry-Cj1oUpLN.d.ts.map → default-registry-Cl3pHo4n.d.ts.map} +1 -1
- package/dist/{eval-campaign-C2k-m4aY.js → eval-campaign-mDKhkdUq.js} +2 -2
- package/dist/{eval-campaign-C2k-m4aY.js.map → eval-campaign-mDKhkdUq.js.map} +1 -1
- package/dist/fuzz.d.ts +1 -1
- package/dist/fuzz.js +1 -1
- package/dist/hosted/index.d.ts +2 -2
- package/dist/{index-B11XCkdf2.d.ts → index-3cdlURSk2.d.ts} +2 -2
- package/dist/{index-B11XCkdf2.d.ts.map → index-3cdlURSk2.d.ts.map} +1 -1
- package/dist/{index-DNXoNaFF.d.ts → index-C2fkZhv_.d.ts} +2 -2
- package/dist/{index-DNXoNaFF.d.ts.map → index-C2fkZhv_.d.ts.map} +1 -1
- package/dist/{index-D_F6VAKe.d.ts → index-CXs7QlR5.d.ts} +3 -3
- package/dist/{index-D_F6VAKe.d.ts.map → index-CXs7QlR5.d.ts.map} +1 -1
- package/dist/{index-VTypFU3t.d.ts → index-FpfWFsKm.d.ts} +7 -7
- package/dist/{index-VTypFU3t.d.ts.map → index-FpfWFsKm.d.ts.map} +1 -1
- package/dist/{index-NPeSWD98.d.ts → index-p2TR_iWJ.d.ts} +5 -5
- package/dist/{index-NPeSWD98.d.ts.map → index-p2TR_iWJ.d.ts.map} +1 -1
- package/dist/index.d.ts +21 -21
- package/dist/index.js +8 -8
- package/dist/{llm-client--GR4JbZE.js → llm-client-BNcP4v08.js} +2 -2
- package/dist/{llm-client--GR4JbZE.js.map → llm-client-BNcP4v08.js.map} +1 -1
- package/dist/{llm-client-B_nIBlYo.d.ts → llm-client-BiK4HW0u.d.ts} +2 -2
- package/dist/{llm-client-B_nIBlYo.d.ts.map → llm-client-BiK4HW0u.d.ts.map} +1 -1
- package/dist/meta-eval/index.d.ts +2 -2
- package/dist/multishot/index.d.ts +1 -1
- package/dist/openapi.json +1 -1
- package/dist/{release-report-Crg9oFJ0.d.ts → release-report-DfmKSIEE.d.ts} +3 -3
- package/dist/{release-report-Crg9oFJ0.d.ts.map → release-report-DfmKSIEE.d.ts.map} +1 -1
- package/dist/{replay-RE97Ckjl.d.ts → replay-BI6CVKkp.d.ts} +2 -2
- package/dist/{replay-RE97Ckjl.d.ts.map → replay-BI6CVKkp.d.ts.map} +1 -1
- package/dist/reporting.d.ts +4 -4
- package/dist/{researcher-Q5rpPqZY.d.ts → researcher-DMimgHtN.d.ts} +4 -4
- package/dist/{researcher-Q5rpPqZY.d.ts.map → researcher-DMimgHtN.d.ts.map} +1 -1
- package/dist/{reward-hacking-CW-3HN0n.d.ts → reward-hacking-D-QqXvg-.d.ts} +2 -2
- package/dist/{reward-hacking-CW-3HN0n.d.ts.map → reward-hacking-D-QqXvg-.d.ts.map} +1 -1
- package/dist/rl.d.ts +5 -5
- package/dist/rl.js +1 -1
- package/dist/rollout/index.d.ts +1 -1
- package/dist/{rubric-predictive-validity-lXLmashy.d.ts → rubric-predictive-validity-C1dCLcvb.d.ts} +2 -2
- package/dist/{rubric-predictive-validity-lXLmashy.d.ts.map → rubric-predictive-validity-C1dCLcvb.d.ts.map} +1 -1
- package/dist/{run-evidence-oByzm-dE.d.ts → run-evidence-DokQtX0-.d.ts} +2 -2
- package/dist/{run-evidence-oByzm-dE.d.ts.map → run-evidence-DokQtX0-.d.ts.map} +1 -1
- package/dist/run-record-CN8Zd21B.js.map +1 -1
- package/dist/{run-record-BJnYdTxO.d.ts → run-record-DcObtIGh.d.ts} +4 -14
- package/dist/run-record-DcObtIGh.d.ts.map +1 -0
- package/dist/{runtime-trajectory-BXxG4lyi.d.ts → runtime-trajectory-BW9Wszb-.d.ts} +2 -2
- package/dist/{runtime-trajectory-BXxG4lyi.d.ts.map → runtime-trajectory-BW9Wszb-.d.ts.map} +1 -1
- package/dist/{semantic-concept-judge-b5m3irbR.js → semantic-concept-judge-DKCtoOz8.js} +4 -4
- package/dist/{semantic-concept-judge-b5m3irbR.js.map → semantic-concept-judge-DKCtoOz8.js.map} +1 -1
- package/dist/{server-m5D9cvnG.js → server-Dc_lsOYd.js} +3 -3
- package/dist/{server-m5D9cvnG.js.map → server-Dc_lsOYd.js.map} +1 -1
- package/dist/{skill-usage-D5mlWdAJ.d.ts → skill-usage-BaaxFSJR.d.ts} +5 -5
- package/dist/{skill-usage-D5mlWdAJ.d.ts.map → skill-usage-BaaxFSJR.d.ts.map} +1 -1
- package/dist/{skillopt-optimization-method-D0MVjJdP.d.ts → skillopt-optimization-method-C9M_lxdo.d.ts} +9 -9
- package/dist/{skillopt-optimization-method-D0MVjJdP.d.ts.map → skillopt-optimization-method-C9M_lxdo.d.ts.map} +1 -1
- package/dist/{skillopt-optimization-method-eOJL2570.js → skillopt-optimization-method-CQlz8GQM.js} +3 -3
- package/dist/{skillopt-optimization-method-eOJL2570.js.map → skillopt-optimization-method-CQlz8GQM.js.map} +1 -1
- package/dist/{statistics-Cmj6nynr.d.ts → statistics-DbvkkDPa.d.ts} +2 -2
- package/dist/{statistics-Cmj6nynr.d.ts.map → statistics-DbvkkDPa.d.ts.map} +1 -1
- package/dist/{summary-report-CWwB_LiV.d.ts → summary-report-DnUcjVpV.d.ts} +2 -2
- package/dist/{summary-report-CWwB_LiV.d.ts.map → summary-report-DnUcjVpV.d.ts.map} +1 -1
- package/dist/traces.d.ts +2 -2
- package/dist/{types-CsD5nTfV.d.ts → types-BokuXvOG.d.ts} +4 -4
- package/dist/{types-CsD5nTfV.d.ts.map → types-BokuXvOG.d.ts.map} +1 -1
- package/dist/{types-DGsxbAEd.d.ts → types-Cc3qbqzj.d.ts} +3 -3
- package/dist/{types-DGsxbAEd.d.ts.map → types-Cc3qbqzj.d.ts.map} +1 -1
- package/dist/wire/index.d.ts +2 -2
- package/dist/wire/index.js +1 -1
- package/package.json +2 -2
- package/dist/cost-ledger-DIgQUFZZ.js.map +0 -1
- package/dist/cost-ledger-Dye6jCgg.d.ts.map +0 -1
- package/dist/default-registry-C-vFCSEc.js.map +0 -1
- package/dist/run-record-BJnYdTxO.d.ts.map +0 -1
package/dist/contract/index.js
CHANGED
|
@@ -1,9 +1,10 @@
|
|
|
1
1
|
import { s as ValidationError } from "../errors-8YnH8WlF.js";
|
|
2
|
-
import { L as createChatClient, t as buildDefaultAnalystRegistry } from "../default-registry-
|
|
2
|
+
import { L as createChatClient, t as buildDefaultAnalystRegistry } from "../default-registry-B1JcpnRv.js";
|
|
3
|
+
import { i as CostLedger } from "../cost-ledger-ZAa_P4r0.js";
|
|
3
4
|
import { LLM_MODEL_ATTR_KEYS, SPAN_KIND_ATTR_KEYS } from "../trace-attributes.js";
|
|
4
5
|
import { b as classifyOtlpSpanRole, x as isOtlpModelCall } from "../tools-BmuN627J.js";
|
|
5
6
|
import { r as mapConcurrentRange } from "../concurrency-MUjT7VjM.js";
|
|
6
|
-
import { B as surfaceContentHash, Ct as llmJudge, M as compareOptimizationMethods, O as composeGate, Q as inMemoryCampaignStorage, S as defaultProductionGate, T as heldoutSignificance, V as surfaceHash, X as createRunCostLedger, Y as runCampaign, Z as fsCampaignStorage, _ as buildEvidenceVector, a as emitLoopProvenance, b as powerPreflight, ct as campaignSplitDigest, d as runImprovementLoop, dt as REFERENCE_EQUIVALENCE_INPUT_LIMITS, ft as REFERENCE_EQUIVALENCE_JUDGE_VERSION, g as gepaOptimizationMethod, j as assertOptimizationResult, k as externalTextOptimizationMethod, mt as runReferenceEquivalenceJudge, o as loopProvenanceArgsFromResult, p as runEval, pt as createReferenceEquivalenceJudge, rt as resolveRunDir, t as skillOptOptimizationMethod, v as paretoPolicy, x as heldOutGate, y as paretoSignificanceGate } from "../skillopt-optimization-method-
|
|
7
|
+
import { B as surfaceContentHash, Ct as llmJudge, M as compareOptimizationMethods, O as composeGate, Q as inMemoryCampaignStorage, S as defaultProductionGate, T as heldoutSignificance, V as surfaceHash, X as createRunCostLedger, Y as runCampaign, Z as fsCampaignStorage, _ as buildEvidenceVector, a as emitLoopProvenance, b as powerPreflight, ct as campaignSplitDigest, d as runImprovementLoop, dt as REFERENCE_EQUIVALENCE_INPUT_LIMITS, ft as REFERENCE_EQUIVALENCE_JUDGE_VERSION, g as gepaOptimizationMethod, j as assertOptimizationResult, k as externalTextOptimizationMethod, mt as runReferenceEquivalenceJudge, o as loopProvenanceArgsFromResult, p as runEval, pt as createReferenceEquivalenceJudge, rt as resolveRunDir, t as skillOptOptimizationMethod, v as paretoPolicy, x as heldOutGate, y as paretoSignificanceGate } from "../skillopt-optimization-method-CQlz8GQM.js";
|
|
7
8
|
import { v as pairedBootstrap } from "../statistics-CnnxdpOg.js";
|
|
8
9
|
import { i as parseRunRecordSafe, r as modelHasSnapshot } from "../run-record-CN8Zd21B.js";
|
|
9
10
|
import { i as summarizeTraceErrors, n as recordAggregateMeasurements, r as summarizeExecutionMeasurements, t as readTaskFailureLabels } from "../task-failure-attributes-CQZlB3et.js";
|
|
@@ -14,7 +15,7 @@ import { a as fromPiSession, c as observeCodeAgentSession, i as fromOpenCodeSess
|
|
|
14
15
|
import { t as createHostedClient } from "../client-CYzbdJOZ.js";
|
|
15
16
|
import { dirname, join } from "node:path";
|
|
16
17
|
import { mkdir, readFile, readdir, stat, writeFile } from "node:fs/promises";
|
|
17
|
-
import { agentCandidateBenchmarkSuiteSchema, agentCandidateBenchmarkTaskSchema, agentCandidateBundleSchema, agentCandidateEvaluationPolicySchema, agentCandidateExperimentSchema, agentImprovementMeasuredComparisonSchema, agentProfileImprovementExperimentSchema, agentProfileImprovementMeasuredComparisonSchema, agentProfileImprovementRunCellSchema, agentProfileImprovementRunReceiptSchema, agentProfileImprovementSuiteInputsSchema, agentProfileImprovementSuiteSchema, agentProfileImprovementTaskSchema, candidateExecutionEvidenceSchema, canonicalCandidateDigest, canonicalCandidateJson, omitTopLevelDigest } from "@tangle-network/agent-interface";
|
|
18
|
+
import { agentCandidateBenchmarkSuiteSchema, agentCandidateBenchmarkTaskSchema, agentCandidateBundleSchema, agentCandidateEvaluationPolicySchema, agentCandidateExperimentSchema, agentImprovementMeasuredComparisonSchema, agentProfileImprovementExperimentSchema, agentProfileImprovementMeasuredComparisonSchema, agentProfileImprovementRunCellSchema, agentProfileImprovementRunReceiptSchema, agentProfileImprovementSuiteInputsSchema, agentProfileImprovementSuiteSchema, agentProfileImprovementTaskSchema, candidateExecutionEvidenceSchema, canonicalCandidateDigest, canonicalCandidateJson, numbersApproximatelyEqual, omitTopLevelDigest } from "@tangle-network/agent-interface";
|
|
18
19
|
//#region src/contract/self-improve.ts
|
|
19
20
|
/**
|
|
20
21
|
* Run one complete improvement job.
|
|
@@ -546,6 +547,19 @@ function requirePositiveInteger(value, field) {
|
|
|
546
547
|
return value;
|
|
547
548
|
}
|
|
548
549
|
//#endregion
|
|
550
|
+
//#region src/contract/fixed-spend.ts
|
|
551
|
+
function addFixedSpend(left, right) {
|
|
552
|
+
return {
|
|
553
|
+
inputTokens: left.inputTokens + right.inputTokens,
|
|
554
|
+
outputTokens: left.outputTokens + right.outputTokens,
|
|
555
|
+
cachedInputTokens: left.cachedInputTokens + right.cachedInputTokens,
|
|
556
|
+
reasoningTokens: left.reasoningTokens + right.reasoningTokens,
|
|
557
|
+
modelCalls: left.modelCalls + right.modelCalls,
|
|
558
|
+
costUsdNanos: left.costUsdNanos + right.costUsdNanos,
|
|
559
|
+
costProvenance: left.costProvenance === "observed" && right.costProvenance === "observed" ? "observed" : "estimated"
|
|
560
|
+
};
|
|
561
|
+
}
|
|
562
|
+
//#endregion
|
|
549
563
|
//#region src/contract/concurrent-map.ts
|
|
550
564
|
/** Map both arms while bounding the actual executions, not the number of pairs. */
|
|
551
565
|
async function mapPairedConcurrent(options) {
|
|
@@ -568,6 +582,62 @@ async function mapPairedConcurrent(options) {
|
|
|
568
582
|
}));
|
|
569
583
|
}
|
|
570
584
|
//#endregion
|
|
585
|
+
//#region src/contract/paid-paired-measurement.ts
|
|
586
|
+
/**
|
|
587
|
+
* Reserve the complete signed suite before its first cell starts.
|
|
588
|
+
*
|
|
589
|
+
* The host executor receives each task's signed cap through `execute`; this
|
|
590
|
+
* batch reservation prevents a complete comparison from turning into a partial
|
|
591
|
+
* one after earlier analysis or search has used the remaining budget.
|
|
592
|
+
*/
|
|
593
|
+
async function runPaidPairedMeasurement(options) {
|
|
594
|
+
assertNonNegativeFinite(options.maximumCostUsd, "maximumCostUsd");
|
|
595
|
+
const costLedger = resolveCostLedger(options);
|
|
596
|
+
const startedAt = performance.now();
|
|
597
|
+
const paid = await costLedger.runPaidCall({
|
|
598
|
+
callId: options.call.callId,
|
|
599
|
+
channel: options.call.channel,
|
|
600
|
+
phase: options.call.phase,
|
|
601
|
+
actor: options.call.actor,
|
|
602
|
+
model: options.call.model,
|
|
603
|
+
maximumCharge: { externallyEnforcedMaximumUsd: options.maximumCostUsd },
|
|
604
|
+
...options.call.tags ? { tags: options.call.tags } : {},
|
|
605
|
+
...options.signal ? { signal: options.signal } : {},
|
|
606
|
+
execute: async (signal) => await mapPairedConcurrent({
|
|
607
|
+
count: options.count,
|
|
608
|
+
maxConcurrency: options.maxConcurrency,
|
|
609
|
+
label: options.label,
|
|
610
|
+
signal,
|
|
611
|
+
map: options.execute
|
|
612
|
+
}),
|
|
613
|
+
receipt: options.receipt
|
|
614
|
+
});
|
|
615
|
+
const wallDurationMs = performance.now() - startedAt;
|
|
616
|
+
if (!paid.succeeded) throw paid.error;
|
|
617
|
+
if (paid.receipt.costUnknown) throw new Error(`${options.label} did not capture the complete measurement cost`);
|
|
618
|
+
return {
|
|
619
|
+
measurements: paid.value,
|
|
620
|
+
wallDurationMs,
|
|
621
|
+
cost: paid.receipt.actualCostUsd === void 0 ? {
|
|
622
|
+
kind: "estimated",
|
|
623
|
+
usd: paid.receipt.costUsd
|
|
624
|
+
} : {
|
|
625
|
+
kind: "observed",
|
|
626
|
+
usd: paid.receipt.costUsd
|
|
627
|
+
}
|
|
628
|
+
};
|
|
629
|
+
}
|
|
630
|
+
function resolveCostLedger(options) {
|
|
631
|
+
if (options.budgetUsd === void 0) return options.costLedger ?? new CostLedger();
|
|
632
|
+
assertNonNegativeFinite(options.budgetUsd, "budgetUsd");
|
|
633
|
+
if (!options.costLedger) throw new Error(`${options.label} with a policy budget requires one shared CostLedger`);
|
|
634
|
+
if (options.costLedger.costCeilingUsd !== options.budgetUsd) throw new Error(`${options.label} CostLedger ceiling must equal the frozen policy budget`);
|
|
635
|
+
return options.costLedger;
|
|
636
|
+
}
|
|
637
|
+
function assertNonNegativeFinite(value, label) {
|
|
638
|
+
if (!Number.isFinite(value) || value < 0) throw new Error(`${label} must be a non-negative finite number`);
|
|
639
|
+
}
|
|
640
|
+
//#endregion
|
|
571
641
|
//#region src/contract/measured-comparison.ts
|
|
572
642
|
/** Content-address one task before any measured execution can see it. */
|
|
573
643
|
function sealCandidateBenchmarkTask(material) {
|
|
@@ -614,12 +684,23 @@ function verifyCandidateExperiment(input) {
|
|
|
614
684
|
async function runCandidateExperiment(options) {
|
|
615
685
|
const experiment = verifyCandidateExperiment(options.experiment);
|
|
616
686
|
const { suite, tasks } = experiment.benchmark;
|
|
617
|
-
|
|
687
|
+
const run = await runPaidPairedMeasurement({
|
|
618
688
|
count: suite.taskDigests.length * suite.reps,
|
|
619
689
|
maxConcurrency: options.maxConcurrency ?? 2,
|
|
620
690
|
label: "candidate experiment",
|
|
691
|
+
budgetUsd: experiment.policy.budgetUsd,
|
|
692
|
+
...options.costLedger ? { costLedger: options.costLedger } : {},
|
|
693
|
+
maximumCostUsd: candidateMeasurementMaximumCostUsd(experiment),
|
|
694
|
+
call: {
|
|
695
|
+
callId: `candidate-measurement:${experiment.digest}`,
|
|
696
|
+
channel: "measurement",
|
|
697
|
+
phase: "heldout",
|
|
698
|
+
actor: experiment.digest,
|
|
699
|
+
model: candidateMeasurementModel(experiment),
|
|
700
|
+
tags: { experimentDigest: experiment.digest }
|
|
701
|
+
},
|
|
621
702
|
...options.signal ? { signal: options.signal } : {},
|
|
622
|
-
async
|
|
703
|
+
async execute(index, arm, signal) {
|
|
623
704
|
const taskIndex = Math.floor(index / suite.reps);
|
|
624
705
|
const repetition = index % suite.reps;
|
|
625
706
|
const task = tasks[taskIndex];
|
|
@@ -630,7 +711,7 @@ async function runCandidateExperiment(options) {
|
|
|
630
711
|
taskIndex,
|
|
631
712
|
repetition
|
|
632
713
|
};
|
|
633
|
-
return options.execute({
|
|
714
|
+
return verifyExecutionEvidence(await options.execute({
|
|
634
715
|
experiment,
|
|
635
716
|
arm,
|
|
636
717
|
bundle: experiment[arm],
|
|
@@ -638,9 +719,22 @@ async function runCandidateExperiment(options) {
|
|
|
638
719
|
benchmarkCell,
|
|
639
720
|
seed,
|
|
640
721
|
signal
|
|
641
|
-
});
|
|
722
|
+
}));
|
|
723
|
+
},
|
|
724
|
+
receipt(measurements) {
|
|
725
|
+
return candidateMeasurementCostReceipt(measurements, candidateMeasurementModel(experiment));
|
|
726
|
+
}
|
|
727
|
+
});
|
|
728
|
+
return {
|
|
729
|
+
measurements: run.measurements.map((measurement, index) => verifyMeasurement(experiment, measurement, index)),
|
|
730
|
+
measurement: {
|
|
731
|
+
wallDurationMs: run.wallDurationMs,
|
|
732
|
+
cost: {
|
|
733
|
+
usd: run.cost.usd,
|
|
734
|
+
provenance: run.cost.kind
|
|
735
|
+
}
|
|
642
736
|
}
|
|
643
|
-
}
|
|
737
|
+
};
|
|
644
738
|
}
|
|
645
739
|
/**
|
|
646
740
|
* Calculate the shared paired decision from any complete receipt shape.
|
|
@@ -651,8 +745,11 @@ async function runCandidateExperiment(options) {
|
|
|
651
745
|
*/
|
|
652
746
|
function evaluatePairedMeasurements(options) {
|
|
653
747
|
if (options.measurements.length === 0) throw new Error("paired measurement evaluation requires at least one paired cell");
|
|
654
|
-
const
|
|
655
|
-
|
|
748
|
+
const preparationCost = options.preparationCost ?? {
|
|
749
|
+
usd: 0,
|
|
750
|
+
provenance: "observed"
|
|
751
|
+
};
|
|
752
|
+
if (!Number.isFinite(preparationCost.usd) || preparationCost.usd < 0) throw new Error("paired measurement evaluation preparation cost must be a non-negative number");
|
|
656
753
|
if (typeof options.sharedScorerChannel !== "boolean") throw new Error("paired measurement evaluation sharedScorerChannel must be a boolean");
|
|
657
754
|
const policy = agentCandidateEvaluationPolicySchema.parse(options.policy);
|
|
658
755
|
const measurements = options.measurements.map((measurement, index) => projectPairedMeasurement(measurement, index, options.adapter));
|
|
@@ -738,12 +835,22 @@ function evaluatePairedMeasurements(options) {
|
|
|
738
835
|
const guardedDimensions = new Set(criticalDimensions);
|
|
739
836
|
const missingCriticalDimensions = criticalDimensions.filter((dimension) => !dimensions.includes(dimension));
|
|
740
837
|
const regressions = objectives.filter((objective) => objective.kind === "dimension" && guardedDimensions.has(objective.name) && objective.availability === "measured" && objective.confidenceInterval.lower < -regressionTolerance);
|
|
741
|
-
const
|
|
742
|
-
const
|
|
838
|
+
const measurementCostUsd = measurements.reduce((sum, measurement) => sum + measurement.baseline.costUsd + measurement.candidate.costUsd, 0);
|
|
839
|
+
const measurementWorkDurationMs = measurements.reduce((sum, measurement) => sum + measurement.baseline.latencyMs + measurement.candidate.latencyMs, 0);
|
|
743
840
|
const incompleteRuns = measurements.flatMap((measurement) => [measurement.baseline, measurement.candidate]).filter((run) => !run.completed);
|
|
744
841
|
const failedCandidateResults = measurements.filter((measurement) => !measurement.candidate.passed);
|
|
745
|
-
const
|
|
746
|
-
|
|
842
|
+
const derivedMeasurementCost = {
|
|
843
|
+
usd: measurementCostUsd,
|
|
844
|
+
provenance: measurements.every((measurement) => measurement.baseline.costProvenance === "observed" && measurement.candidate.costProvenance === "observed") ? "observed" : "estimated"
|
|
845
|
+
};
|
|
846
|
+
const measurementCost = options.measurementCost ?? derivedMeasurementCost;
|
|
847
|
+
assertKnownCost(measurementCost, "paired measurement cost");
|
|
848
|
+
if (options.measurementCost !== void 0 && (measurementCost.provenance !== derivedMeasurementCost.provenance || !numbersApproximatelyEqual(measurementCost.usd, derivedMeasurementCost.usd))) throw new Error("paired measurement cost does not match its signed receipts");
|
|
849
|
+
const totalCost = {
|
|
850
|
+
usd: preparationCost.usd + measurementCost.usd,
|
|
851
|
+
provenance: preparationCost.provenance === "observed" && measurementCost.provenance === "observed" ? "observed" : "estimated"
|
|
852
|
+
};
|
|
853
|
+
const budgetPassed = budgetUsd === void 0 || totalCost.usd < budgetUsd || numbersApproximatelyEqual(totalCost.usd, budgetUsd);
|
|
747
854
|
const checks = [
|
|
748
855
|
{
|
|
749
856
|
name: "paired-significance",
|
|
@@ -778,7 +885,7 @@ function evaluatePairedMeasurements(options) {
|
|
|
778
885
|
...missingCriticalDimensions.length === 0 ? [] : [`critical dimensions missing: ${missingCriticalDimensions.join(", ")}`],
|
|
779
886
|
...incompleteRuns.length === 0 ? [] : [`${incompleteRuns.length} benchmark executions did not exit successfully`],
|
|
780
887
|
...failedCandidateResults.length === 0 ? [] : [`candidate failed ${failedCandidateResults.length} benchmark tasks`],
|
|
781
|
-
...budgetPassed ? [] : [`total cost ${
|
|
888
|
+
...budgetPassed ? [] : [`total cost ${totalCost.usd} exceeded budget ${budgetUsd}`]
|
|
782
889
|
];
|
|
783
890
|
return {
|
|
784
891
|
overall: {
|
|
@@ -802,9 +909,9 @@ function evaluatePairedMeasurements(options) {
|
|
|
802
909
|
sharedScorerChannel: options.sharedScorerChannel,
|
|
803
910
|
reason: power.reason
|
|
804
911
|
},
|
|
805
|
-
|
|
806
|
-
|
|
807
|
-
|
|
912
|
+
measurementCost,
|
|
913
|
+
totalCost,
|
|
914
|
+
measurementWorkDurationMs
|
|
808
915
|
};
|
|
809
916
|
}
|
|
810
917
|
/** Build the only publishable comparison: paired statistics over Runtime receipts. */
|
|
@@ -815,7 +922,8 @@ function measuredComparisonFromCandidateExperiment(options) {
|
|
|
815
922
|
if (measurements.length !== expectedN) throw new Error(`candidate experiment is incomplete (${measurements.length}/${expectedN} paired cells)`);
|
|
816
923
|
verifyStableProfileMaterialization(measurements);
|
|
817
924
|
if (!options.runId.trim()) throw new Error("candidate experiment runId is required");
|
|
818
|
-
|
|
925
|
+
assertPhaseAccounting(options.preparation, "candidate preparation");
|
|
926
|
+
assertPhaseAccounting(options.measurement, "candidate measurement");
|
|
819
927
|
const evaluation = evaluatePairedMeasurements({
|
|
820
928
|
measurements: measurements.map((measurement, index) => ({
|
|
821
929
|
cellId: cellIds(experiment)[index],
|
|
@@ -824,12 +932,10 @@ function measuredComparisonFromCandidateExperiment(options) {
|
|
|
824
932
|
policy: experiment.policy,
|
|
825
933
|
adapter: candidateExecutionEvidenceAdapter,
|
|
826
934
|
sharedScorerChannel: true,
|
|
827
|
-
|
|
935
|
+
preparationCost: options.preparation.cost,
|
|
936
|
+
measurementCost: options.measurement.cost
|
|
828
937
|
});
|
|
829
938
|
const diff = deriveCandidateBundleDiff(experiment);
|
|
830
|
-
const searchDurationMs = options.searchDurationMs ?? 0;
|
|
831
|
-
const totalCostUsd = evaluation.totalCostUsd;
|
|
832
|
-
const durationMs = evaluation.executionDurationMs + searchDurationMs;
|
|
833
939
|
const provisional = agentImprovementMeasuredComparisonSchema.parse({
|
|
834
940
|
kind: "agent-improvement-measured-comparison",
|
|
835
941
|
experiment,
|
|
@@ -850,12 +956,16 @@ function measuredComparisonFromCandidateExperiment(options) {
|
|
|
850
956
|
diff,
|
|
851
957
|
evaluation: {
|
|
852
958
|
generationsExplored: options.generationsExplored ?? 0,
|
|
853
|
-
|
|
854
|
-
|
|
855
|
-
|
|
856
|
-
|
|
857
|
-
|
|
858
|
-
|
|
959
|
+
preparation: options.preparation,
|
|
960
|
+
measurement: {
|
|
961
|
+
wallDurationMs: options.measurement.wallDurationMs,
|
|
962
|
+
workDurationMs: evaluation.measurementWorkDurationMs,
|
|
963
|
+
cost: evaluation.measurementCost
|
|
964
|
+
},
|
|
965
|
+
total: {
|
|
966
|
+
wallDurationMs: options.preparation.wallDurationMs + options.measurement.wallDurationMs,
|
|
967
|
+
cost: evaluation.totalCost
|
|
968
|
+
}
|
|
859
969
|
},
|
|
860
970
|
...options.metadata ? { metadata: options.metadata } : {}
|
|
861
971
|
});
|
|
@@ -880,12 +990,52 @@ function verifyCandidateExperimentComparison(input) {
|
|
|
880
990
|
runId: comparison.provenance.runId,
|
|
881
991
|
...comparison.candidate ? { candidate: comparison.candidate } : {},
|
|
882
992
|
generationsExplored: comparison.evaluation.generationsExplored,
|
|
883
|
-
|
|
884
|
-
|
|
993
|
+
preparation: comparison.evaluation.preparation,
|
|
994
|
+
measurement: {
|
|
995
|
+
wallDurationMs: comparison.evaluation.measurement.wallDurationMs,
|
|
996
|
+
cost: comparison.evaluation.measurement.cost
|
|
997
|
+
},
|
|
885
998
|
...comparison.metadata ? { metadata: comparison.metadata } : {}
|
|
886
999
|
})) !== canonicalCandidateDigest(comparison)) throw new Error("candidate experiment comparison does not match its Runtime receipts");
|
|
887
1000
|
return comparison;
|
|
888
1001
|
}
|
|
1002
|
+
function candidateMeasurementMaximumCostUsd(experiment) {
|
|
1003
|
+
const maximum = experiment.benchmark.tasks.reduce((sum, task) => sum + task.limits.maxCostUsd * experiment.benchmark.suite.reps * 2, 0);
|
|
1004
|
+
if (!Number.isFinite(maximum) || maximum < 0) throw new Error("candidate measurement maximum cost is invalid");
|
|
1005
|
+
return maximum;
|
|
1006
|
+
}
|
|
1007
|
+
function candidateMeasurementModel(experiment) {
|
|
1008
|
+
const models = [...new Set(experiment.benchmark.tasks.map((task) => task.model.model))];
|
|
1009
|
+
return models.length === 1 ? models[0] : "multiple-models";
|
|
1010
|
+
}
|
|
1011
|
+
function candidateMeasurementCostReceipt(measurements, model) {
|
|
1012
|
+
const usage = measurements.reduce((sum, measurement) => addFixedSpend(addFixedSpend(sum, combinedUsage(measurement.baseline)), combinedUsage(measurement.candidate)), {
|
|
1013
|
+
inputTokens: 0,
|
|
1014
|
+
outputTokens: 0,
|
|
1015
|
+
cachedInputTokens: 0,
|
|
1016
|
+
reasoningTokens: 0,
|
|
1017
|
+
modelCalls: 0,
|
|
1018
|
+
costUsdNanos: 0,
|
|
1019
|
+
costProvenance: "observed"
|
|
1020
|
+
});
|
|
1021
|
+
const costUsd = usage.costUsdNanos / 1e9;
|
|
1022
|
+
return {
|
|
1023
|
+
model,
|
|
1024
|
+
inputTokens: usage.inputTokens,
|
|
1025
|
+
outputTokens: usage.outputTokens,
|
|
1026
|
+
cachedTokens: usage.cachedInputTokens,
|
|
1027
|
+
reasoningTokens: usage.reasoningTokens,
|
|
1028
|
+
...usage.costProvenance === "observed" ? { actualCostUsd: costUsd } : { estimatedCostUsd: costUsd }
|
|
1029
|
+
};
|
|
1030
|
+
}
|
|
1031
|
+
function assertPhaseAccounting(accounting, label) {
|
|
1032
|
+
if (!Number.isFinite(accounting.wallDurationMs) || accounting.wallDurationMs < 0) throw new Error(`${label} wall duration must be a non-negative number`);
|
|
1033
|
+
assertKnownCost(accounting.cost, `${label} cost`);
|
|
1034
|
+
}
|
|
1035
|
+
function assertKnownCost(cost, label) {
|
|
1036
|
+
if (!Number.isFinite(cost.usd) || cost.usd < 0) throw new Error(`${label} must be a non-negative number`);
|
|
1037
|
+
if (cost.provenance !== "observed" && cost.provenance !== "estimated") throw new Error(`${label} provenance must be observed or estimated`);
|
|
1038
|
+
}
|
|
889
1039
|
function deriveCandidateBundleDiff(experiment) {
|
|
890
1040
|
const changed = [
|
|
891
1041
|
"profile",
|
|
@@ -1061,10 +1211,13 @@ function projectRun(run, adapter, label) {
|
|
|
1061
1211
|
const completed = adapter.completed(run);
|
|
1062
1212
|
const passed = adapter.passed(run);
|
|
1063
1213
|
if (typeof completed !== "boolean" || typeof passed !== "boolean") throw new Error(`${label} completion and pass values must be booleans`);
|
|
1214
|
+
const costProvenance = adapter.costProvenance(run);
|
|
1215
|
+
if (costProvenance !== "observed" && costProvenance !== "estimated") throw new Error(`${label} cost provenance must be observed or estimated`);
|
|
1064
1216
|
return {
|
|
1065
1217
|
score: finiteMeasurement(adapter.score(run), `${label} score`),
|
|
1066
1218
|
dimensions,
|
|
1067
1219
|
costUsd: nonNegativeMeasurement(adapter.costUsd(run), `${label} cost`),
|
|
1220
|
+
costProvenance,
|
|
1068
1221
|
latencyMs: nonNegativeMeasurement(adapter.latencyMs(run), `${label} latency`),
|
|
1069
1222
|
completed,
|
|
1070
1223
|
passed
|
|
@@ -1149,6 +1302,7 @@ const candidateExecutionEvidenceAdapter = {
|
|
|
1149
1302
|
score: (evidence) => evidence.receipt.benchmarkResult.material.score,
|
|
1150
1303
|
dimensions: (evidence) => evidence.receipt.benchmarkResult.material.dimensions,
|
|
1151
1304
|
costUsd: costFromEvidence,
|
|
1305
|
+
costProvenance: (evidence) => combinedUsage(evidence).costProvenance,
|
|
1152
1306
|
latencyMs: latencyFromEvidence,
|
|
1153
1307
|
completed: completedSuccessfully,
|
|
1154
1308
|
passed: (evidence) => evidence.receipt.benchmarkResult.material.passed
|
|
@@ -1162,14 +1316,7 @@ function latencyFromEvidence(evidence) {
|
|
|
1162
1316
|
function combinedUsage(evidence) {
|
|
1163
1317
|
const candidate = evidence.receipt.modelSettlement.material.usage;
|
|
1164
1318
|
const grader = evidence.receipt.benchmarkResult.material.grading.usage;
|
|
1165
|
-
return
|
|
1166
|
-
inputTokens: candidate.inputTokens + grader.inputTokens,
|
|
1167
|
-
outputTokens: candidate.outputTokens + grader.outputTokens,
|
|
1168
|
-
cachedInputTokens: candidate.cachedInputTokens + grader.cachedInputTokens,
|
|
1169
|
-
reasoningTokens: candidate.reasoningTokens + grader.reasoningTokens,
|
|
1170
|
-
modelCalls: candidate.modelCalls + grader.modelCalls,
|
|
1171
|
-
costUsdNanos: candidate.costUsdNanos + grader.costUsdNanos
|
|
1172
|
-
};
|
|
1319
|
+
return addFixedSpend(candidate, grader);
|
|
1173
1320
|
}
|
|
1174
1321
|
function cellIds(experiment) {
|
|
1175
1322
|
const { suite, tasks } = experiment.benchmark;
|
|
@@ -1235,15 +1382,45 @@ function verifyAgentProfileImprovementExperiment(input) {
|
|
|
1235
1382
|
*/
|
|
1236
1383
|
async function runAgentProfileImprovementExperiment(options) {
|
|
1237
1384
|
const experiment = verifyAgentProfileImprovementExperiment(options.experiment);
|
|
1238
|
-
|
|
1385
|
+
const run = await runPaidPairedMeasurement({
|
|
1239
1386
|
count: experiment.benchmark.suite.taskDigests.length * experiment.benchmark.suite.reps,
|
|
1240
1387
|
maxConcurrency: options.maxConcurrency ?? 2,
|
|
1241
1388
|
label: "profile improvement experiment",
|
|
1389
|
+
budgetUsd: experiment.policy.budgetUsd,
|
|
1390
|
+
...options.costLedger ? { costLedger: options.costLedger } : {},
|
|
1391
|
+
maximumCostUsd: profileMeasurementMaximumCostUsd(experiment),
|
|
1392
|
+
call: {
|
|
1393
|
+
callId: `profile-improvement-measurement:${experiment.digest}`,
|
|
1394
|
+
channel: "measurement",
|
|
1395
|
+
phase: "heldout",
|
|
1396
|
+
actor: experiment.executionRef.identity,
|
|
1397
|
+
model: profileMeasurementModel(experiment),
|
|
1398
|
+
tags: {
|
|
1399
|
+
experimentDigest: experiment.digest,
|
|
1400
|
+
executorDigest: experiment.executionRef.digest
|
|
1401
|
+
}
|
|
1402
|
+
},
|
|
1242
1403
|
...options.signal ? { signal: options.signal } : {},
|
|
1243
|
-
|
|
1244
|
-
|
|
1404
|
+
async execute(index, arm, signal) {
|
|
1405
|
+
const expected = profileExecutionInput(experiment, arm, index, signal);
|
|
1406
|
+
const receipt = agentProfileImprovementRunReceiptSchema.parse(await options.execute(expected));
|
|
1407
|
+
verifyProfileReceiptContract(receipt, expected, index);
|
|
1408
|
+
return receipt;
|
|
1409
|
+
},
|
|
1410
|
+
receipt(measurements) {
|
|
1411
|
+
return profileMeasurementCostReceipt(measurements, profileMeasurementModel(experiment));
|
|
1245
1412
|
}
|
|
1246
|
-
})
|
|
1413
|
+
});
|
|
1414
|
+
return {
|
|
1415
|
+
measurements: run.measurements.map((measurement, index) => verifyProfileMeasurement(experiment, measurement, index)),
|
|
1416
|
+
measurement: {
|
|
1417
|
+
wallDurationMs: run.wallDurationMs,
|
|
1418
|
+
cost: {
|
|
1419
|
+
usd: run.cost.usd,
|
|
1420
|
+
provenance: run.cost.kind
|
|
1421
|
+
}
|
|
1422
|
+
}
|
|
1423
|
+
};
|
|
1247
1424
|
}
|
|
1248
1425
|
/** Build the only publishable profile comparison from complete host receipts. */
|
|
1249
1426
|
function measuredComparisonFromAgentProfileImprovementExperiment(options) {
|
|
@@ -1252,8 +1429,6 @@ function measuredComparisonFromAgentProfileImprovementExperiment(options) {
|
|
|
1252
1429
|
const expectedCount = experiment.benchmark.suite.taskDigests.length * experiment.benchmark.suite.reps;
|
|
1253
1430
|
if (measurements.length !== expectedCount) throw new Error(`profile improvement experiment is incomplete (${measurements.length}/${expectedCount} paired cells)`);
|
|
1254
1431
|
if (!options.runId.trim()) throw new Error("profile improvement experiment runId is required");
|
|
1255
|
-
const searchCostUsd = nonNegative(options.searchCostUsd ?? 0, "searchCostUsd");
|
|
1256
|
-
const searchDurationMs = nonNegative(options.searchDurationMs ?? 0, "searchDurationMs");
|
|
1257
1432
|
const evaluation = evaluatePairedMeasurements({
|
|
1258
1433
|
measurements: measurements.map((measurement, index) => ({
|
|
1259
1434
|
cellId: profileCellId(experiment, index),
|
|
@@ -1262,7 +1437,8 @@ function measuredComparisonFromAgentProfileImprovementExperiment(options) {
|
|
|
1262
1437
|
policy: experiment.policy,
|
|
1263
1438
|
adapter: profileReceiptAdapter,
|
|
1264
1439
|
sharedScorerChannel: true,
|
|
1265
|
-
|
|
1440
|
+
preparationCost: options.preparation.cost,
|
|
1441
|
+
measurementCost: options.measurement.cost
|
|
1266
1442
|
});
|
|
1267
1443
|
const provisional = agentProfileImprovementMeasuredComparisonSchema.parse({
|
|
1268
1444
|
kind: "agent-profile-improvement-measured-comparison",
|
|
@@ -1284,12 +1460,16 @@ function measuredComparisonFromAgentProfileImprovementExperiment(options) {
|
|
|
1284
1460
|
diff: canonicalCandidateJson(experiment.change),
|
|
1285
1461
|
evaluation: {
|
|
1286
1462
|
generationsExplored: options.generationsExplored ?? 0,
|
|
1287
|
-
|
|
1288
|
-
|
|
1289
|
-
|
|
1290
|
-
|
|
1291
|
-
|
|
1292
|
-
|
|
1463
|
+
preparation: options.preparation,
|
|
1464
|
+
measurement: {
|
|
1465
|
+
wallDurationMs: options.measurement.wallDurationMs,
|
|
1466
|
+
workDurationMs: evaluation.measurementWorkDurationMs,
|
|
1467
|
+
cost: evaluation.measurementCost
|
|
1468
|
+
},
|
|
1469
|
+
total: {
|
|
1470
|
+
wallDurationMs: options.preparation.wallDurationMs + options.measurement.wallDurationMs,
|
|
1471
|
+
cost: evaluation.totalCost
|
|
1472
|
+
}
|
|
1293
1473
|
},
|
|
1294
1474
|
...options.metadata ? { metadata: options.metadata } : {}
|
|
1295
1475
|
});
|
|
@@ -1314,12 +1494,48 @@ function verifyAgentProfileImprovementExperimentComparison(input) {
|
|
|
1314
1494
|
runId: comparison.provenance.runId,
|
|
1315
1495
|
...comparison.candidate ? { candidate: comparison.candidate } : {},
|
|
1316
1496
|
generationsExplored: comparison.evaluation.generationsExplored,
|
|
1317
|
-
|
|
1318
|
-
|
|
1497
|
+
preparation: comparison.evaluation.preparation,
|
|
1498
|
+
measurement: {
|
|
1499
|
+
wallDurationMs: comparison.evaluation.measurement.wallDurationMs,
|
|
1500
|
+
cost: comparison.evaluation.measurement.cost
|
|
1501
|
+
},
|
|
1319
1502
|
...comparison.metadata ? { metadata: comparison.metadata } : {}
|
|
1320
1503
|
})) !== canonicalCandidateDigest(comparison)) throw new Error("profile improvement comparison does not match its Runtime receipts");
|
|
1321
1504
|
return comparison;
|
|
1322
1505
|
}
|
|
1506
|
+
function profileMeasurementMaximumCostUsd(experiment) {
|
|
1507
|
+
const pairs = experiment.benchmark.suite.reps;
|
|
1508
|
+
const maximum = experiment.benchmark.tasks.reduce((sum, task) => sum + task.limits.maxCostUsd * pairs * 2, 0);
|
|
1509
|
+
if (!Number.isFinite(maximum) || maximum < 0) throw new Error("profile improvement measurement maximum cost is invalid");
|
|
1510
|
+
return maximum;
|
|
1511
|
+
}
|
|
1512
|
+
function profileMeasurementModel(experiment) {
|
|
1513
|
+
const models = [...new Set(experiment.benchmark.tasks.map((task) => task.model.model))];
|
|
1514
|
+
return models.length === 1 ? models[0] : "multiple-models";
|
|
1515
|
+
}
|
|
1516
|
+
function profileMeasurementCostReceipt(measurements, model) {
|
|
1517
|
+
const usage = measurements.reduce((sum, measurement) => addFixedSpend(addFixedSpend(sum, combinedProfileUsage(measurement.baseline)), combinedProfileUsage(measurement.candidate)), {
|
|
1518
|
+
inputTokens: 0,
|
|
1519
|
+
outputTokens: 0,
|
|
1520
|
+
cachedInputTokens: 0,
|
|
1521
|
+
reasoningTokens: 0,
|
|
1522
|
+
modelCalls: 0,
|
|
1523
|
+
costUsdNanos: 0,
|
|
1524
|
+
costProvenance: "observed"
|
|
1525
|
+
});
|
|
1526
|
+
const costUsd = usage.costUsdNanos / 1e9;
|
|
1527
|
+
return {
|
|
1528
|
+
model,
|
|
1529
|
+
inputTokens: usage.inputTokens,
|
|
1530
|
+
outputTokens: usage.outputTokens,
|
|
1531
|
+
cachedTokens: usage.cachedInputTokens,
|
|
1532
|
+
reasoningTokens: usage.reasoningTokens,
|
|
1533
|
+
...usage.costProvenance === "observed" ? { actualCostUsd: costUsd } : { estimatedCostUsd: costUsd }
|
|
1534
|
+
};
|
|
1535
|
+
}
|
|
1536
|
+
function combinedProfileUsage(receipt) {
|
|
1537
|
+
return addFixedSpend(receipt.usage, receipt.grading.usage);
|
|
1538
|
+
}
|
|
1323
1539
|
function profileExecutionInput(experiment, arm, index, signal) {
|
|
1324
1540
|
const { task, taskIndex, repetition, seed } = profileCell(experiment, index);
|
|
1325
1541
|
const stateDigest = experiment[arm].stateDigest;
|
|
@@ -1356,17 +1572,18 @@ function verifyProfileMeasurement(experiment, input, index) {
|
|
|
1356
1572
|
const baseline = agentProfileImprovementRunReceiptSchema.parse(material.baseline);
|
|
1357
1573
|
const candidate = agentProfileImprovementRunReceiptSchema.parse(material.candidate);
|
|
1358
1574
|
if (baseline.runCell.digest !== expectedBaseline.runCell.digest || candidate.runCell.digest !== expectedCandidate.runCell.digest) throw new Error(`profile improvement measurement ${index} substituted a measured arm`);
|
|
1359
|
-
|
|
1360
|
-
|
|
1575
|
+
verifyProfileReceiptContract(baseline, expectedBaseline, index);
|
|
1576
|
+
verifyProfileReceiptContract(candidate, expectedCandidate, index);
|
|
1361
1577
|
if (baseline.executionId === candidate.executionId || baseline.digest === candidate.digest) throw new Error(`profile improvement measurement ${index} reused one execution across arms`);
|
|
1362
1578
|
return {
|
|
1363
1579
|
baseline,
|
|
1364
1580
|
candidate
|
|
1365
1581
|
};
|
|
1366
1582
|
}
|
|
1367
|
-
function
|
|
1583
|
+
function verifyProfileReceiptContract(receipt, expected, index) {
|
|
1368
1584
|
const task = expected.task;
|
|
1369
1585
|
if (canonicalCandidateDigest(receipt.resolvedModel) !== canonicalCandidateDigest(task.model) || canonicalCandidateDigest(receipt.limits) !== canonicalCandidateDigest(task.limits) || canonicalCandidateDigest(receipt.grading.grader) !== canonicalCandidateDigest(task.grader)) throw new Error(`profile improvement measurement ${index} substituted its ${expected.arm} task contract`);
|
|
1586
|
+
if (canonicalCandidateDigest(receipt.executionRef) !== canonicalCandidateDigest(expected.experiment.executionRef)) throw new Error(`profile improvement measurement ${index} substituted its ${expected.arm} executor`);
|
|
1370
1587
|
}
|
|
1371
1588
|
function profileCell(experiment, index) {
|
|
1372
1589
|
const { suite, tasks } = experiment.benchmark;
|
|
@@ -1389,14 +1606,11 @@ const profileReceiptAdapter = {
|
|
|
1389
1606
|
score: (receipt) => receipt.grading.score,
|
|
1390
1607
|
dimensions: (receipt) => receipt.grading.dimensions,
|
|
1391
1608
|
costUsd: (receipt) => (receipt.usage.costUsdNanos + receipt.grading.usage.costUsdNanos) / 1e9,
|
|
1609
|
+
costProvenance: (receipt) => receipt.usage.costProvenance === "observed" && receipt.grading.usage.costProvenance === "observed" ? "observed" : "estimated",
|
|
1392
1610
|
latencyMs: (receipt) => receipt.timing.durationMs + receipt.grading.timing.durationMs,
|
|
1393
1611
|
completed: (receipt) => receipt.outcome.status === "succeeded",
|
|
1394
1612
|
passed: (receipt) => receipt.grading.passed
|
|
1395
1613
|
};
|
|
1396
|
-
function nonNegative(value, label) {
|
|
1397
|
-
if (!Number.isFinite(value) || value < 0) throw new Error(`${label} must be a non-negative number`);
|
|
1398
|
-
return value;
|
|
1399
|
-
}
|
|
1400
1614
|
//#endregion
|
|
1401
1615
|
//#region src/contract/intake/run-record-dir.ts
|
|
1402
1616
|
/**
|