@tangle-network/agent-eval 0.145.6 → 0.145.8
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +19 -0
- package/README.md +4 -0
- package/dist/analyst/index.d.ts +10 -10
- package/dist/analyst/index.js +1 -1
- package/dist/{backend-integrity-CsVin_Wb.d.ts → backend-integrity-BffcHGdm.d.ts} +2 -2
- package/dist/{backend-integrity-CsVin_Wb.d.ts.map → backend-integrity-BffcHGdm.d.ts.map} +1 -1
- package/dist/{benchmark-Ceoan7vk.d.ts → benchmark-DbiZbPdH.d.ts} +3 -3
- package/dist/{benchmark-Ceoan7vk.d.ts.map → benchmark-DbiZbPdH.d.ts.map} +1 -1
- package/dist/{benchmark-command-DspwA7cv.js → benchmark-command-rWP4-dJ6.js} +2 -2
- package/dist/{benchmark-command-DspwA7cv.js.map → benchmark-command-rWP4-dJ6.js.map} +1 -1
- package/dist/benchmarks/index.d.ts +5 -5
- package/dist/benchmarks/index.js +2 -2
- package/dist/campaign/index.d.ts +8 -8
- package/dist/campaign/index.js +4 -4
- package/dist/{campaign-jTOvqnse.js → campaign-Dg2B4W6d.js} +35 -11
- package/dist/campaign-Dg2B4W6d.js.map +1 -0
- package/dist/{capture-fetch-DDvpjVRU.d.ts → capture-fetch-hykwHfDI.d.ts} +2 -2
- package/dist/{capture-fetch-DDvpjVRU.d.ts.map → capture-fetch-hykwHfDI.d.ts.map} +1 -1
- package/dist/cli.js +1 -1
- package/dist/{client-DCVe0CwG.d.ts → client-D_TIV9pJ.d.ts} +4 -4
- package/dist/{client-DCVe0CwG.d.ts.map → client-D_TIV9pJ.d.ts.map} +1 -1
- package/dist/contract/index.d.ts +11 -11
- package/dist/contract/index.js +4 -4
- package/dist/{default-registry-CgzxnJsj.d.ts → default-registry-cpHtP2Og.d.ts} +6 -6
- package/dist/{default-registry-CgzxnJsj.d.ts.map → default-registry-cpHtP2Og.d.ts.map} +1 -1
- package/dist/{define-agent-eval-DLhZ8IHi.d.ts → define-agent-eval-BgG5DIOS.d.ts} +6 -6
- package/dist/{define-agent-eval-DLhZ8IHi.d.ts.map → define-agent-eval-BgG5DIOS.d.ts.map} +1 -1
- package/dist/{define-agent-eval-CViDh2P9.js → define-agent-eval-cSMEFzVu.js} +4 -4
- package/dist/{define-agent-eval-CViDh2P9.js.map → define-agent-eval-cSMEFzVu.js.map} +1 -1
- package/dist/{engine-FRZ3RLvT.d.ts → engine-DJqRKbhs.d.ts} +7 -7
- package/dist/{engine-FRZ3RLvT.d.ts.map → engine-DJqRKbhs.d.ts.map} +1 -1
- package/dist/{eval-campaign-UB-usSQ2.js → eval-campaign-aNqpefCS.js} +2 -2
- package/dist/{eval-campaign-UB-usSQ2.js.map → eval-campaign-aNqpefCS.js.map} +1 -1
- package/dist/{exact-types-BH1twmAJ.d.ts → exact-types-CzbhVDr2.d.ts} +2 -2
- package/dist/{exact-types-BH1twmAJ.d.ts.map → exact-types-CzbhVDr2.d.ts.map} +1 -1
- package/dist/experiment/index.d.ts +3 -3
- package/dist/{feedback-trajectory-juOozjAc.d.ts → feedback-trajectory-D9vYSob_.d.ts} +3 -3
- package/dist/{feedback-trajectory-juOozjAc.d.ts.map → feedback-trajectory-D9vYSob_.d.ts.map} +1 -1
- package/dist/hosted/index.d.ts +3 -3
- package/dist/index-BKShPTwZ.d.ts +1 -0
- package/dist/{index-COQYtuRF.d.ts → index-BNPtkBPf.d.ts} +2 -2
- package/dist/{index-COQYtuRF.d.ts.map → index-BNPtkBPf.d.ts.map} +1 -1
- package/dist/{index-DF3ynqmJ.d.ts → index-CM-e2LiV.d.ts} +11 -11
- package/dist/{index-DF3ynqmJ.d.ts.map → index-CM-e2LiV.d.ts.map} +1 -1
- package/dist/{index-PgqfwhsM.d.ts → index-YrUFx3FU.d.ts} +5 -5
- package/dist/{index-PgqfwhsM.d.ts.map → index-YrUFx3FU.d.ts.map} +1 -1
- package/dist/index.d.ts +22 -22
- package/dist/index.js +9 -9
- package/dist/{insight-report-CRi-Ufrj.d.ts → insight-report-BeT8KCgI.d.ts} +3 -3
- package/dist/{insight-report-CRi-Ufrj.d.ts.map → insight-report-BeT8KCgI.d.ts.map} +1 -1
- package/dist/{llm-judge-DuYa4SEA.js → llm-judge-DhiJqSjB.js} +3 -3
- package/dist/{llm-judge-DuYa4SEA.js.map → llm-judge-DhiJqSjB.js.map} +1 -1
- package/dist/meta-eval/index.d.ts +1 -1
- package/dist/{mint-Dj9Ww_3I.js → mint-BV6tLVWl.js} +2 -2
- package/dist/{mint-Dj9Ww_3I.js.map → mint-BV6tLVWl.js.map} +1 -1
- package/dist/multishot/index.d.ts +1 -1
- package/dist/openapi.json +1 -1
- package/dist/{pre-registration-DeRvl9sE.d.ts → pre-registration-zFSLEiFU.d.ts} +2 -2
- package/dist/{pre-registration-DeRvl9sE.d.ts.map → pre-registration-zFSLEiFU.d.ts.map} +1 -1
- package/dist/{produced-state-D6j7qy1Q.js → produced-state-BNyyud4g.js} +2 -2
- package/dist/{produced-state-D6j7qy1Q.js.map → produced-state-BNyyud4g.js.map} +1 -1
- package/dist/{promotion-policy-CsMZJOB-.d.ts → promotion-policy-u3wj6w3U.d.ts} +2 -2
- package/dist/{promotion-policy-CsMZJOB-.d.ts.map → promotion-policy-u3wj6w3U.d.ts.map} +1 -1
- package/dist/{provenance-8L-_4xiL.d.ts → provenance-AACfvbUR.d.ts} +124 -108
- package/dist/provenance-AACfvbUR.d.ts.map +1 -0
- package/dist/{registry-oJeeI4-a.d.ts → registry-B_1Frl8a.d.ts} +3 -3
- package/dist/{registry-oJeeI4-a.d.ts.map → registry-B_1Frl8a.d.ts.map} +1 -1
- package/dist/{release-confidence-BFRE5WSp.d.ts → release-confidence-4XrqlpFD.d.ts} +3 -3
- package/dist/{release-confidence-BFRE5WSp.d.ts.map → release-confidence-4XrqlpFD.d.ts.map} +1 -1
- package/dist/{release-confidence-CxDuiAev.js → release-confidence-BknrpBnO.js} +2 -2
- package/dist/{release-confidence-CxDuiAev.js.map → release-confidence-BknrpBnO.js.map} +1 -1
- package/dist/reporting.d.ts +3 -3
- package/dist/reporting.js +1 -1
- package/dist/{researcher-DaN4GST-.d.ts → researcher-Du-oniHp.d.ts} +3 -3
- package/dist/{researcher-DaN4GST-.d.ts.map → researcher-Du-oniHp.d.ts.map} +1 -1
- package/dist/{reward-hacking-DSSTuI9r.d.ts → reward-hacking-Bu8ev6PR.d.ts} +2 -2
- package/dist/{reward-hacking-DSSTuI9r.d.ts.map → reward-hacking-Bu8ev6PR.d.ts.map} +1 -1
- package/dist/{reward-hacking-DNgjilrV.js → reward-hacking-DFo2FU5J.js} +2 -2
- package/dist/{reward-hacking-DNgjilrV.js.map → reward-hacking-DFo2FU5J.js.map} +1 -1
- package/dist/rl.d.ts +5 -5
- package/dist/rl.js +4 -4
- package/dist/rollout/index.d.ts +1 -1
- package/dist/rollout/index.js +2 -2
- package/dist/{rollout-BWtw0I_6.js → rollout-ytVQ7WT8.js} +2 -2
- package/dist/{rollout-BWtw0I_6.js.map → rollout-ytVQ7WT8.js.map} +1 -1
- package/dist/{rubric-predictive-validity-BgxtKe4G.d.ts → rubric-predictive-validity-C7LnNvF2.d.ts} +2 -2
- package/dist/{rubric-predictive-validity-BgxtKe4G.d.ts.map → rubric-predictive-validity-C7LnNvF2.d.ts.map} +1 -1
- package/dist/{run-record-BvHPVS-i.js → run-record-D2lDdSAz.js} +5 -3
- package/dist/run-record-D2lDdSAz.js.map +1 -0
- package/dist/{run-record-CKiihE6f.d.ts → run-record-DVV82Gwh.d.ts} +8 -3
- package/dist/run-record-DVV82Gwh.d.ts.map +1 -0
- package/dist/{skillopt-optimization-method-BTls2l-Z.js → skillopt-optimization-method-C6JSfYxT.js} +2 -2
- package/dist/{skillopt-optimization-method-BTls2l-Z.js.map → skillopt-optimization-method-C6JSfYxT.js.map} +1 -1
- package/dist/{skillopt-optimization-method-CFDSzD-7.d.ts → skillopt-optimization-method-DO6OFgcz.d.ts} +5 -5
- package/dist/{skillopt-optimization-method-CFDSzD-7.d.ts.map → skillopt-optimization-method-DO6OFgcz.d.ts.map} +1 -1
- package/dist/{statistical-heldout-C4De2tRI.d.ts → statistical-heldout-CBGPSZ5X.d.ts} +3 -3
- package/dist/{statistical-heldout-C4De2tRI.d.ts.map → statistical-heldout-CBGPSZ5X.d.ts.map} +1 -1
- package/dist/{store-tool-spans-CggeC1LB.d.ts → store-tool-spans-C9c4R7ca.d.ts} +4 -4
- package/dist/{store-tool-spans-CggeC1LB.d.ts.map → store-tool-spans-C9c4R7ca.d.ts.map} +1 -1
- package/dist/{summary-report-CaL-Hnxt.d.ts → summary-report-B__Y5ub3.d.ts} +2 -2
- package/dist/{summary-report-CaL-Hnxt.d.ts.map → summary-report-B__Y5ub3.d.ts.map} +1 -1
- package/dist/{tool-groups-BdcoEgtV.d.ts → tool-groups-CZotz-e_.d.ts} +3 -3
- package/dist/tool-groups-CZotz-e_.d.ts.map +1 -0
- package/dist/trace-repair/index.d.ts +2 -2
- package/dist/traces.d.ts +6 -6
- package/dist/traces.js +1 -1
- package/dist/{types-BxLccGMf.d.ts → types-BjsNDR49.d.ts} +3 -3
- package/dist/{types-BxLccGMf.d.ts.map → types-BjsNDR49.d.ts.map} +1 -1
- package/dist/{types-BRsxjg7z.d.ts → types-C4bSVIr7.d.ts} +3 -3
- package/dist/{types-BRsxjg7z.d.ts.map → types-C4bSVIr7.d.ts.map} +1 -1
- package/dist/{types-DLQx4mKU.d.ts → types-vXyshMwx.d.ts} +2 -2
- package/dist/{types-DLQx4mKU.d.ts.map → types-vXyshMwx.d.ts.map} +1 -1
- package/dist/wire/index.d.ts +1 -1
- package/docs/eval-surface-map.md +2 -1
- package/docs/wire-protocol.md +4 -0
- package/package.json +1 -1
- package/dist/campaign-jTOvqnse.js.map +0 -1
- package/dist/index-Ba3YrbAL.d.ts +0 -1
- package/dist/provenance-8L-_4xiL.d.ts.map +0 -1
- package/dist/run-record-BvHPVS-i.js.map +0 -1
- package/dist/run-record-CKiihE6f.d.ts.map +0 -1
- package/dist/tool-groups-BdcoEgtV.d.ts.map +0 -1
|
@@ -1,8 +1,8 @@
|
|
|
1
|
-
import { s as RunSplitTag } from "../run-record-
|
|
2
|
-
import { a as CampaignResult } from "../types-
|
|
3
|
-
import { a as BenchmarkFamily, c as BenchmarkSource, i as BenchmarkEvaluation, l as BenchmarkTaskKind, n as BenchmarkAdapter, o as BenchmarkResponder, r as BenchmarkDatasetItem, s as BenchmarkScenario, t as BENCHMARK_SPLIT_SEED, u as deterministicSplit } from "../types-
|
|
4
|
-
import {
|
|
5
|
-
import "../index-
|
|
1
|
+
import { s as RunSplitTag } from "../run-record-DVV82Gwh.js";
|
|
2
|
+
import { a as CampaignResult } from "../types-BjsNDR49.js";
|
|
3
|
+
import { a as BenchmarkFamily, c as BenchmarkSource, i as BenchmarkEvaluation, l as BenchmarkTaskKind, n as BenchmarkAdapter, o as BenchmarkResponder, r as BenchmarkDatasetItem, s as BenchmarkScenario, t as BENCHMARK_SPLIT_SEED, u as deterministicSplit } from "../types-C4bSVIr7.js";
|
|
4
|
+
import { Tt as CampaignStorage } from "../provenance-AACfvbUR.js";
|
|
5
|
+
import "../index-CM-e2LiV.js";
|
|
6
6
|
//#region src/benchmarks/calibration.d.ts
|
|
7
7
|
interface BenchmarkMetricCalibrationOptions<TPayload = unknown, TArtifact = string> {
|
|
8
8
|
adapter: BenchmarkAdapter<BenchmarkDatasetItem<TPayload>, TPayload, TArtifact>;
|
package/dist/benchmarks/index.js
CHANGED
|
@@ -1,7 +1,7 @@
|
|
|
1
1
|
import { t as __exportAll } from "../rolldown-runtime-8H4AJuhK.js";
|
|
2
|
-
import { K as runCampaign } from "../llm-judge-
|
|
2
|
+
import { K as runCampaign } from "../llm-judge-DhiJqSjB.js";
|
|
3
3
|
import { x as fsCampaignStorage } from "../external-optimizer-subprocess-BhKYK0Jv.js";
|
|
4
|
-
import "../campaign-
|
|
4
|
+
import "../campaign-Dg2B4W6d.js";
|
|
5
5
|
import { join } from "node:path";
|
|
6
6
|
//#region src/benchmarks/calibration.ts
|
|
7
7
|
async function calibrateBenchmarkMetric(options) {
|
package/dist/campaign/index.d.ts
CHANGED
|
@@ -1,11 +1,11 @@
|
|
|
1
1
|
import { c as CostLedgerHandle, w as PendingCostCallView } from "../cost-ledger-DbQdN3nO.js";
|
|
2
|
-
import { _ as ProposalFinding, v as ProposalFindingOrigin, x as makeProposalFinding } from "../types-
|
|
2
|
+
import { _ as ProposalFinding, v as ProposalFindingOrigin, x as makeProposalFinding } from "../types-vXyshMwx.js";
|
|
3
3
|
import { a as ExternalOptimizerEndpointFormat, c as ExternalOptimizerModelBudget, d as ExternalOptimizerModelCallResult, f as ExternalOptimizerModelExecutionObservation, g as ExternalTextCandidate, h as ExternalOptimizerRunnerCommand, i as ExternalOptimizerChatRequest, l as ExternalOptimizerModelCall, n as DEFAULT_EXTERNAL_OPTIMIZER_PROCESS_LIMITS, o as ExternalOptimizerEvaluationObservation, p as ExternalOptimizerProcessLimits, r as ExternalOptimizerCallbackLimits, s as ExternalOptimizerEvaluationRefusalReason, t as DEFAULT_EXTERNAL_OPTIMIZER_CALLBACK_LIMITS, u as ExternalOptimizerModelCallRequest, v as resolveExternalOptimizerCallbackLimits, y as resolveExternalOptimizerProcessLimits } from "../external-optimizer-contracts-lixrOZdX.js";
|
|
4
|
-
import { A as LabeledScenarioWrite, B as ScoredSurfaceOutcome, C as JudgeDimension, D as LabeledScenarioSampleArgs, E as LabeledScenarioRecord, F as ProposeContext, G as labelTrustRank, H as SurfaceProposer, I as ProposedCandidate, L as RedactionStatus, M as OptimizerConfig, N as ParetoParent, O as LabeledScenarioSource, P as ProposalTrackContext, R as Scenario, S as JudgeConfig, T as LabelTrust, U as TraceSpan, V as SessionScript, W as isProposedCandidate, _ as GateDecision, a as CampaignResult, b as GenerationRecord, c as CampaignTraceWriter, d as DispatchContext, f as DispatchFn, g as GateContribution, h as GateContext, i as CampaignCostMeter, j as MutableSurface, k as LabeledScenarioStore, l as CodeSurface, m as GateCheckStatus, n as CampaignArtifactWriter, o as CampaignScenarioIdentity, p as Gate, r as CampaignCellResult, s as CampaignTokenUsage, t as CampaignAggregates, u as ComponentSurface, v as GateResult, w as JudgeScore, x as JudgeAggregate, y as GenerationCandidate, z as ScenarioAggregate } from "../types-
|
|
5
|
-
import { $ as OptimizationTokenUsage, At as
|
|
6
|
-
import { A as campaignScenarioIdentity, C as ExternalTextOptimizerContext, D as decodeExternalTextCandidate, E as ExternalTextEvaluationResponse, I as ReferenceEquivalenceJudgeOptions, M as campaignSplitDigestFromIdentities, O as assertCampaignDesign, R as ReferenceEquivalenceScenario, S as ExternalTextOptimizationMethodConfig, T as ExternalOptimizationExample, _ as DefaultProductionGateOptions, a as GepaAdaptiveEngineRun, b as composeGate, c as GepaOptimizationMethodConfig, d as gepaOptimizationMethod, f as OpenAICompatibleOptimizerModel, g as DefaultProductionGateCheck, h as heldOutGate, i as skillOptOptimizationMethod, j as campaignSplitDigest, k as assertCampaignSplitIdentity, l as GepaOptimizationRecipe, m as HeldOutGateOptions, n as SkillOptRunnerCommand, o as GepaEngineOptions, p as OptimizerModelBudget, r as SkillOptTrainerConfig, s as GepaEngineRun, t as SkillOptOptimizationMethodConfig, u as GepaRunnerCommand, v as DefaultProductionRewardHackingOptions, w as ExternalTextOptimizerResult, x as externalTextOptimizationMethod, y as defaultProductionGate, z as createReferenceEquivalenceJudge } from "../skillopt-optimization-method-
|
|
7
|
-
import { $ as SearchPlannedEvent, $n as CrossSurfacePairEvidence, $t as RunProfileMatrixResult, A as SearchAccountingAudit, An as CrossSurfaceAdditionRejectionReason, At as ProfileMatrixSegmentResult, B as SearchCostAccounting, Bn as CrossSurfaceComponentEvidence, Bt as ScoreboardRow, C as surfaceHash, Cn as discoverEvalFixtures, Ct as tangleTracesRoot, D as FileSearchLedger, Dn as analyzeCrossSurfaceInteractions, Dt as ProfileMatrixCoverage, E as acquireSingleRunLock, En as planEvalFixtureRun, Et as FinalizedProfileMatrixResult, F as SearchCandidateRegisteredEvent, Fn as CrossSurfaceCandidateComparison, Ft as runProfileMatrixSegment, G as SearchLedgerEvent, Gn as CrossSurfaceIneligibilityReason, Gt as renderScoreboardMarkdown, H as SearchLedger, Hn as CrossSurfaceDistribution, Ht as UserStory, I as SearchCandidateSlot, In as CrossSurfaceCandidateEvidence, It as PlaybackContext, J as SearchLedgerTrustedHeadMode, Jn as CrossSurfaceInteractionPath, Jt as userStoryScoreboard, K as SearchLedgerHash, Kn as CrossSurfaceInteractionAwareSelection, Kt as scoreUserStory, L as SearchCandidateSlotClosedEvent, Ln as CrossSurfaceCandidateOutcome, Lt as PlaybackDriver, M as SearchAttemptAccounting, Mn as CrossSurfaceBestSingleSelection, Mt as SegmentedProfileMatrixError, N as SearchCandidateDecidedEvent, Nn as CrossSurfaceBootstrapPolicy, Nt as createProfileMatrixPlan, O as OpenSearchLedgerOptions, On as AnalyzeCrossSurfaceInteractionsInput, Ot as ProfileMatrixPlan, P as SearchCandidateLineage, Pn as CrossSurfaceCandidate, Pt as finalizeProfileMatrix, Q as SearchPlan, Qn as CrossSurfacePairCompatibility, Qt as RunProfileMatrixOptions, R as SearchCandidateSurface, Rn as CrossSurfaceCandidateSummary, Rt as PlaybackStep, S as surfaceContentHash, Sn as PlanEvalFixtureRunOptions, St as resolveRunDir, T as SingleRunLockOptions, Tn as loadEvalFixtureScenarios, Tt as FinalizeProfileMatrixOptions, U as SearchLedgerAppendResult, Un as CrossSurfaceEligibility, Ut as UserStoryVerdict, V as SearchFailureReason, Vn as CrossSurfaceCompositionStep, Vt as ScoreboardSummary, W as SearchLedgerEntry, Wn as CrossSurfaceEvidenceBreakdown, Wt as makePlaybackDispatch, X as SearchOperationKind, Xn as CrossSurfaceInteractionTask, Xt as ProfileMatrixError, Y as SearchModelIdentity, Yn as CrossSurfaceInteractionReport, Yt as ProfileDispatchFn, Z as SearchOperationRecordedEvent, Zn as CrossSurfaceNaiveStackSelection, Zt as ProfileSummary, _ as assertCodeSurfaceIdentity, _n as EvalFixtureLoadOptions, _t as compareRankKeys, a as WorktreeAdapterError, an as LabeledScenarioStoreError, ar as CrossSurfaceSelections, at as SearchSurfaceKind, b as componentSurfaceIdentityMaterial, bn as EvalFixtureValidationMode, bt as scoreDiscrimination, c as verifyCodeSurface, cn as RolloutCall, cr as TraceAnalystArtifact, ct as SearchTokenAccounting, d as PhoenixEvaluationResultLike, dn as classifyUngroundedLiterals, dr as traceAnalystQualityJudge, dt as SearchLedgerConflictError, en as ScenarioRollup, er as CrossSurfacePairIncompatibilityReason, et as SearchPlannedOperation, f as PhoenixEvaluatorLike, fn as rolloutArgumentDiff, ft as SearchLedgerError, g as isTransientTransportFailure, gn as EvalFixtureFile, gt as campaignMeanComposite, h as TransientFailureOptions, hn as EvalFixture, ht as campaignBreakdown, i as WorktreeAdapter, in as FsLabeledScenarioStoreOptions, ir as CrossSurfaceSelectionPolicy, it as SearchSurfaceEvidence, j as SearchArtifactRef, jn as CrossSurfaceAttemptCompleteness, jt as RunProfileMatrixSegmentOptions, k as SEARCH_LEDGER_SCHEMA, kn as CrossSurfaceAdditionDecision, kt as ProfileMatrixRow, l as AutoevalsScoreLike, ln as ScoredRollout, lr as TraceAnalystScenario, lt as openSearchLedger, m as phoenixEvaluatorJudge, mn as neutralizationGate, mt as CampaignBreakdown, n as GitWorktreeAdapterOptions, nn as neutralizeText, nr as CrossSurfaceRankedSingle, nt as SearchSourceRef, o as gitWorktreeAdapter, on as RolloutArgumentDiff, or as CrossSurfaceTaskRow, ot as SearchTaskAttemptedEvent, p as autoevalsScorerJudge, pn as NeutralizationGateOptions, pt as SearchLedgerIntegrityError, q as SearchLedgerReplay, qn as CrossSurfaceInteractionEffect, qt as scoreboardSummary, r as Worktree, rn as FsLabeledScenarioStore, rr as CrossSurfaceRelativeCost, rt as SearchSurfaceEffect, s as resolveWorktreePath, sn as RolloutArgumentDiffOptions, sr as BuildTraceAnalystSurfaceDispatchOptions, st as SearchTaskOutcome, t as CodeSurfaceVerification, tn as runProfileMatrix, tr as CrossSurfacePairwiseEntry, tt as SearchPlannedTask, u as AutoevalsScorerLike, un as UngroundedLiteralReport, ur as buildTraceAnalystSurfaceDispatch, ut as validateSearchLedgerEvent, v as assertComponentSurface, vn as EvalFixtureRunPlan, vt as DiscriminationScore, w as SingleRunLock, wn as loadEvalFixture, wt as CreateProfileMatrixPlanOptions, x as renderSurfaceDiff, xn as LoadEvalFixtureScenariosOptions, xt as selectDiscriminative, y as codeSurfaceIdentityMaterial, yn as EvalFixtureScenario, yt as ScenarioSignal, z as SearchCompletedEvent, zn as CrossSurfaceComponent, zt as ScoreboardRenderOptions } from "../index-
|
|
4
|
+
import { A as LabeledScenarioWrite, B as ScoredSurfaceOutcome, C as JudgeDimension, D as LabeledScenarioSampleArgs, E as LabeledScenarioRecord, F as ProposeContext, G as labelTrustRank, H as SurfaceProposer, I as ProposedCandidate, L as RedactionStatus, M as OptimizerConfig, N as ParetoParent, O as LabeledScenarioSource, P as ProposalTrackContext, R as Scenario, S as JudgeConfig, T as LabelTrust, U as TraceSpan, V as SessionScript, W as isProposedCandidate, _ as GateDecision, a as CampaignResult, b as GenerationRecord, c as CampaignTraceWriter, d as DispatchContext, f as DispatchFn, g as GateContribution, h as GateContext, i as CampaignCostMeter, j as MutableSurface, k as LabeledScenarioStore, l as CodeSurface, m as GateCheckStatus, n as CampaignArtifactWriter, o as CampaignScenarioIdentity, p as Gate, r as CampaignCellResult, s as CampaignTokenUsage, t as CampaignAggregates, u as ComponentSurface, v as GateResult, w as JudgeScore, x as JudgeAggregate, y as GenerationCandidate, z as ScenarioAggregate } from "../types-BjsNDR49.js";
|
|
5
|
+
import { $ as OptimizationTokenUsage, At as OpenAutoPrResult, C as RunOptimizationOptions, Ct as PlanCampaignRunOptions, D as runEval, Dt as fsCampaignStorage, E as RunEvalOptions, Et as createRunCostLedger, G as OptimizationMethodComparison, H as CompareOptimizationMethodsOptions, J as OptimizationMethodProvenance, K as OptimizationMethodInput, Mt as LlmJudgeDimension, Nt as LlmJudgeOptions, Ot as inMemoryCampaignStorage, Pt as llmJudge, Q as OptimizationPackageSource, S as PremeasuredOptimizationBaseline, St as CampaignRunPlanCell, T as runOptimization, Tt as CampaignStorage, U as ComparisonCost, W as OptimizationMethod, X as OptimizationMethodRunOptions, Y as OptimizationMethodResult, Z as OptimizationMethodScore, _ as provenanceSpansPath, _t as cellCachePath, a as LoopProvenanceBackend, at as GepaCandidatePopulationCandidate, b as RunImprovementLoopResult, bt as runCampaign, c as LoopProvenanceOptimizationMethod, ct as readGepaCandidatePopulationArtifact, d as campaignMeasurementDigest, dt as ExternalOptimizerObservationSummary, et as combineComparisonCosts, f as canonicalDigest, ft as ExternalOptimizerSubmittedCandidate, g as provenanceRecordPath, gt as readCachedCell, h as loopProvenanceSpans, ht as CacheRead, i as LoopProvenanceArgsFromResult, it as GepaCandidatePopulationArtifact, jt as openAutoPr, kt as OpenAutoPrOptions, l as LoopProvenanceRecord, lt as ExternalOptimizerExecutionSummary, m as loopProvenanceArgsFromResult, mt as CacheIssueReason, n as EmitLoopProvenanceArgs, nt as costFromLedgerSummary, o as LoopProvenanceCandidate, ot as GepaCandidatePopulationSummary, p as emitLoopProvenance, pt as readExternalOptimizerObservationArtifact, q as OptimizationMethodPairwise, r as EmitLoopProvenanceResult, rt as optimizationTokenUsageFromSummary, s as LoopProvenanceEvidence, st as GepaCandidateSelectionScore, t as BuildLoopProvenanceArgs, tt as compareOptimizationMethods, u as buildLoopProvenanceRecord, ut as ExternalOptimizerObservationArtifact, v as verifyLoopProvenanceRecord, vt as CampaignCellFailureReceipt, w as RunOptimizationResult, wt as planCampaignRun, x as runImprovementLoop, xt as CampaignRunPlan, y as RunImprovementLoopOptions, yt as RunCampaignOptions } from "../provenance-AACfvbUR.js";
|
|
6
|
+
import { A as campaignScenarioIdentity, C as ExternalTextOptimizerContext, D as decodeExternalTextCandidate, E as ExternalTextEvaluationResponse, I as ReferenceEquivalenceJudgeOptions, M as campaignSplitDigestFromIdentities, O as assertCampaignDesign, R as ReferenceEquivalenceScenario, S as ExternalTextOptimizationMethodConfig, T as ExternalOptimizationExample, _ as DefaultProductionGateOptions, a as GepaAdaptiveEngineRun, b as composeGate, c as GepaOptimizationMethodConfig, d as gepaOptimizationMethod, f as OpenAICompatibleOptimizerModel, g as DefaultProductionGateCheck, h as heldOutGate, i as skillOptOptimizationMethod, j as campaignSplitDigest, k as assertCampaignSplitIdentity, l as GepaOptimizationRecipe, m as HeldOutGateOptions, n as SkillOptRunnerCommand, o as GepaEngineOptions, p as OptimizerModelBudget, r as SkillOptTrainerConfig, s as GepaEngineRun, t as SkillOptOptimizationMethodConfig, u as GepaRunnerCommand, v as DefaultProductionRewardHackingOptions, w as ExternalTextOptimizerResult, x as externalTextOptimizationMethod, y as defaultProductionGate, z as createReferenceEquivalenceJudge } from "../skillopt-optimization-method-DO6OFgcz.js";
|
|
7
|
+
import { $ as SearchPlannedEvent, $n as CrossSurfacePairEvidence, $t as RunProfileMatrixResult, A as SearchAccountingAudit, An as CrossSurfaceAdditionRejectionReason, At as ProfileMatrixSegmentResult, B as SearchCostAccounting, Bn as CrossSurfaceComponentEvidence, Bt as ScoreboardRow, C as surfaceHash, Cn as discoverEvalFixtures, Ct as tangleTracesRoot, D as FileSearchLedger, Dn as analyzeCrossSurfaceInteractions, Dt as ProfileMatrixCoverage, E as acquireSingleRunLock, En as planEvalFixtureRun, Et as FinalizedProfileMatrixResult, F as SearchCandidateRegisteredEvent, Fn as CrossSurfaceCandidateComparison, Ft as runProfileMatrixSegment, G as SearchLedgerEvent, Gn as CrossSurfaceIneligibilityReason, Gt as renderScoreboardMarkdown, H as SearchLedger, Hn as CrossSurfaceDistribution, Ht as UserStory, I as SearchCandidateSlot, In as CrossSurfaceCandidateEvidence, It as PlaybackContext, J as SearchLedgerTrustedHeadMode, Jn as CrossSurfaceInteractionPath, Jt as userStoryScoreboard, K as SearchLedgerHash, Kn as CrossSurfaceInteractionAwareSelection, Kt as scoreUserStory, L as SearchCandidateSlotClosedEvent, Ln as CrossSurfaceCandidateOutcome, Lt as PlaybackDriver, M as SearchAttemptAccounting, Mn as CrossSurfaceBestSingleSelection, Mt as SegmentedProfileMatrixError, N as SearchCandidateDecidedEvent, Nn as CrossSurfaceBootstrapPolicy, Nt as createProfileMatrixPlan, O as OpenSearchLedgerOptions, On as AnalyzeCrossSurfaceInteractionsInput, Ot as ProfileMatrixPlan, P as SearchCandidateLineage, Pn as CrossSurfaceCandidate, Pt as finalizeProfileMatrix, Q as SearchPlan, Qn as CrossSurfacePairCompatibility, Qt as RunProfileMatrixOptions, R as SearchCandidateSurface, Rn as CrossSurfaceCandidateSummary, Rt as PlaybackStep, S as surfaceContentHash, Sn as PlanEvalFixtureRunOptions, St as resolveRunDir, T as SingleRunLockOptions, Tn as loadEvalFixtureScenarios, Tt as FinalizeProfileMatrixOptions, U as SearchLedgerAppendResult, Un as CrossSurfaceEligibility, Ut as UserStoryVerdict, V as SearchFailureReason, Vn as CrossSurfaceCompositionStep, Vt as ScoreboardSummary, W as SearchLedgerEntry, Wn as CrossSurfaceEvidenceBreakdown, Wt as makePlaybackDispatch, X as SearchOperationKind, Xn as CrossSurfaceInteractionTask, Xt as ProfileMatrixError, Y as SearchModelIdentity, Yn as CrossSurfaceInteractionReport, Yt as ProfileDispatchFn, Z as SearchOperationRecordedEvent, Zn as CrossSurfaceNaiveStackSelection, Zt as ProfileSummary, _ as assertCodeSurfaceIdentity, _n as EvalFixtureLoadOptions, _t as compareRankKeys, a as WorktreeAdapterError, an as LabeledScenarioStoreError, ar as CrossSurfaceSelections, at as SearchSurfaceKind, b as componentSurfaceIdentityMaterial, bn as EvalFixtureValidationMode, bt as scoreDiscrimination, c as verifyCodeSurface, cn as RolloutCall, cr as TraceAnalystArtifact, ct as SearchTokenAccounting, d as PhoenixEvaluationResultLike, dn as classifyUngroundedLiterals, dr as traceAnalystQualityJudge, dt as SearchLedgerConflictError, en as ScenarioRollup, er as CrossSurfacePairIncompatibilityReason, et as SearchPlannedOperation, f as PhoenixEvaluatorLike, fn as rolloutArgumentDiff, ft as SearchLedgerError, g as isTransientTransportFailure, gn as EvalFixtureFile, gt as campaignMeanComposite, h as TransientFailureOptions, hn as EvalFixture, ht as campaignBreakdown, i as WorktreeAdapter, in as FsLabeledScenarioStoreOptions, ir as CrossSurfaceSelectionPolicy, it as SearchSurfaceEvidence, j as SearchArtifactRef, jn as CrossSurfaceAttemptCompleteness, jt as RunProfileMatrixSegmentOptions, k as SEARCH_LEDGER_SCHEMA, kn as CrossSurfaceAdditionDecision, kt as ProfileMatrixRow, l as AutoevalsScoreLike, ln as ScoredRollout, lr as TraceAnalystScenario, lt as openSearchLedger, m as phoenixEvaluatorJudge, mn as neutralizationGate, mt as CampaignBreakdown, n as GitWorktreeAdapterOptions, nn as neutralizeText, nr as CrossSurfaceRankedSingle, nt as SearchSourceRef, o as gitWorktreeAdapter, on as RolloutArgumentDiff, or as CrossSurfaceTaskRow, ot as SearchTaskAttemptedEvent, p as autoevalsScorerJudge, pn as NeutralizationGateOptions, pt as SearchLedgerIntegrityError, q as SearchLedgerReplay, qn as CrossSurfaceInteractionEffect, qt as scoreboardSummary, r as Worktree, rn as FsLabeledScenarioStore, rr as CrossSurfaceRelativeCost, rt as SearchSurfaceEffect, s as resolveWorktreePath, sn as RolloutArgumentDiffOptions, sr as BuildTraceAnalystSurfaceDispatchOptions, st as SearchTaskOutcome, t as CodeSurfaceVerification, tn as runProfileMatrix, tr as CrossSurfacePairwiseEntry, tt as SearchPlannedTask, u as AutoevalsScorerLike, un as UngroundedLiteralReport, ur as buildTraceAnalystSurfaceDispatch, ut as validateSearchLedgerEvent, v as assertComponentSurface, vn as EvalFixtureRunPlan, vt as DiscriminationScore, w as SingleRunLock, wn as loadEvalFixture, wt as CreateProfileMatrixPlanOptions, x as renderSurfaceDiff, xn as LoadEvalFixtureScenariosOptions, xt as selectDiscriminative, y as codeSurfaceIdentityMaterial, yn as EvalFixtureScenario, yt as ScenarioSignal, z as SearchCompletedEvent, zn as CrossSurfaceComponent, zt as ScoreboardRenderOptions } from "../index-CM-e2LiV.js";
|
|
8
8
|
import { c as powerPreflight, o as PowerPreflight, s as PowerPreflightOptions } from "../pareto-BqNW3LJR.js";
|
|
9
|
-
import { a as ObjectiveSource, c as PromotionPolicy, d as paretoSignificanceGate, i as EvidenceVector, l as buildEvidenceVector, n as AxisVerdict, o as ParetoSignificanceGateOptions, r as BuildEvidenceVectorOptions, s as PromotionObjective, t as AxisEvidence, u as paretoPolicy } from "../promotion-policy-
|
|
10
|
-
import { a as detectScale, c as pairHoldout, d as SequentialDecision, f as SequentialObservation, g as sequentialPairedGate, h as sequentialDecide, i as PairedHoldout, l as SequentialDecideFn, m as SequentialPairedGateOptions, n as HeldoutSignificance, o as dimensionRegressions, p as SequentialPairedGate, r as HeldoutSignificanceOptions, s as heldoutSignificance, t as DimensionRegression, u as SequentialDecideOptions } from "../statistical-heldout-
|
|
11
|
-
export { type AnalyzeCrossSurfaceInteractionsInput, type AutoevalsScoreLike, type AutoevalsScorerLike, type AxisEvidence, type AxisVerdict, type BuildEvidenceVectorOptions, type BuildLoopProvenanceArgs, type BuildTraceAnalystSurfaceDispatchOptions, type CampaignAggregates, type CampaignArtifactWriter, type CampaignBreakdown, type CampaignCellFailureReceipt, type CampaignCellResult, type CampaignCostMeter, type CampaignResult, type CampaignRunPlan, type CampaignRunPlanCell, type CampaignScenarioIdentity, type CampaignStorage, type CampaignTokenUsage, type CampaignTraceWriter, type CodeSurface, type CodeSurfaceVerification, type CompareOptimizationMethodsOptions, type ComparisonCost, type ComponentSurface, type CostLedgerHandle, type CreateProfileMatrixPlanOptions, type CrossSurfaceAdditionDecision, type CrossSurfaceAdditionRejectionReason, type CrossSurfaceAttemptCompleteness, type CrossSurfaceBestSingleSelection, type CrossSurfaceBootstrapPolicy, type CrossSurfaceCandidate, type CrossSurfaceCandidateComparison, type CrossSurfaceCandidateEvidence, type CrossSurfaceCandidateOutcome, type CrossSurfaceCandidateSummary, type CrossSurfaceComponent, type CrossSurfaceComponentEvidence, type CrossSurfaceCompositionStep, type CrossSurfaceDistribution, type CrossSurfaceEligibility, type CrossSurfaceEvidenceBreakdown, type CrossSurfaceIneligibilityReason, type CrossSurfaceInteractionAwareSelection, type CrossSurfaceInteractionEffect, type CrossSurfaceInteractionPath, type CrossSurfaceInteractionReport, type CrossSurfaceInteractionTask, type CrossSurfaceNaiveStackSelection, type CrossSurfacePairCompatibility, type CrossSurfacePairEvidence, type CrossSurfacePairIncompatibilityReason, type CrossSurfacePairwiseEntry, type CrossSurfaceRankedSingle, type CrossSurfaceRelativeCost, type CrossSurfaceSelectionPolicy, type CrossSurfaceSelections, type CrossSurfaceTaskRow, DEFAULT_EXTERNAL_OPTIMIZER_CALLBACK_LIMITS, DEFAULT_EXTERNAL_OPTIMIZER_PROCESS_LIMITS, type DefaultProductionGateCheck, type DefaultProductionGateOptions, type DefaultProductionRewardHackingOptions, type DimensionRegression, type DiscriminationScore, type DispatchContext, type DispatchFn, type EmitLoopProvenanceArgs, type EmitLoopProvenanceResult, type EvalFixture, type EvalFixtureFile, type EvalFixtureLoadOptions, type EvalFixtureRunPlan, type EvalFixtureScenario, type EvalFixtureValidationMode, type EvidenceVector, type ExternalOptimizationExample, type ExternalOptimizerCallbackLimits, type ExternalOptimizerChatRequest, type ExternalOptimizerEndpointFormat, type ExternalOptimizerEvaluationObservation, type ExternalOptimizerEvaluationRefusalReason, type ExternalOptimizerExecutionSummary, type ExternalOptimizerModelBudget, type ExternalOptimizerModelCall, type ExternalOptimizerModelCallRequest, type ExternalOptimizerModelCallResult, type ExternalOptimizerModelExecutionObservation, type ExternalOptimizerObservationArtifact, type ExternalOptimizerObservationSummary, type ExternalOptimizerProcessLimits, type ExternalOptimizerRunnerCommand, type ExternalOptimizerSubmittedCandidate, type ExternalTextCandidate, type ExternalTextEvaluationResponse, type ExternalTextOptimizationMethodConfig, type ExternalTextOptimizerContext, type ExternalTextOptimizerResult, FileSearchLedger, type FinalizeProfileMatrixOptions, type FinalizedProfileMatrixResult, FsLabeledScenarioStore, type FsLabeledScenarioStoreOptions, type Gate, type GateCheckStatus, type GateContext, type GateContribution, type GateDecision, type GateResult, type GenerationCandidate, type GenerationRecord, type GepaAdaptiveEngineRun, type GepaCandidatePopulationArtifact, type GepaCandidatePopulationCandidate, type GepaCandidatePopulationSummary, type GepaCandidateSelectionScore, type GepaEngineOptions, type GepaEngineRun, type GepaOptimizationMethodConfig, type GepaOptimizationRecipe, type GepaRunnerCommand, type GitWorktreeAdapterOptions, type HeldOutGateOptions, type HeldoutSignificance, type HeldoutSignificanceOptions, type JudgeAggregate, type JudgeConfig, type JudgeDimension, type JudgeScore, type LabelTrust, type LabeledScenarioRecord, type LabeledScenarioSampleArgs, type LabeledScenarioSource, type LabeledScenarioStore, LabeledScenarioStoreError, type LabeledScenarioWrite, type LlmJudgeDimension, type LlmJudgeOptions, type LoadEvalFixtureScenariosOptions, type LoopProvenanceArgsFromResult, type LoopProvenanceBackend, type LoopProvenanceCandidate, type LoopProvenanceEvidence, type LoopProvenanceOptimizationMethod, type LoopProvenanceRecord, type MutableSurface, type NeutralizationGateOptions, type ObjectiveSource, type OpenAICompatibleOptimizerModel, type OpenAutoPrOptions, type OpenAutoPrResult, type OpenSearchLedgerOptions, type OptimizationMethod, type OptimizationMethodComparison, type OptimizationMethodInput, type OptimizationMethodPairwise, type OptimizationMethodProvenance, type OptimizationMethodResult, type OptimizationMethodRunOptions, type OptimizationMethodScore, type OptimizationPackageSource, type OptimizationTokenUsage, type OptimizerConfig, type OptimizerModelBudget, type PairedHoldout, type ParetoParent, type ParetoSignificanceGateOptions, type PendingCostCallView, type PhoenixEvaluationResultLike, type PhoenixEvaluatorLike, type PlanCampaignRunOptions, type PlanEvalFixtureRunOptions, type PlaybackContext, type PlaybackDriver, type PlaybackStep, type PowerPreflight, type PowerPreflightOptions, type PremeasuredOptimizationBaseline, type ProfileDispatchFn, type ProfileMatrixCoverage, ProfileMatrixError, type ProfileMatrixPlan, type ProfileMatrixRow, type ProfileMatrixSegmentResult, type ProfileSummary, type PromotionObjective, type PromotionPolicy, type ProposalFinding, type ProposalFindingOrigin, type ProposalTrackContext, type ProposeContext, type ProposedCandidate, type RedactionStatus, type ReferenceEquivalenceJudgeOptions, type ReferenceEquivalenceScenario, type RolloutArgumentDiff, type RolloutArgumentDiffOptions, type RolloutCall, type RunCampaignOptions, type RunEvalOptions, type RunImprovementLoopOptions, type RunImprovementLoopResult, type RunOptimizationOptions, type RunOptimizationResult, type RunProfileMatrixOptions, type RunProfileMatrixResult, type RunProfileMatrixSegmentOptions, SEARCH_LEDGER_SCHEMA, type Scenario, type ScenarioAggregate, type ScenarioRollup, type ScenarioSignal, type ScoreboardRenderOptions, type ScoreboardRow, type ScoreboardSummary, type ScoredRollout, type ScoredSurfaceOutcome, type SearchAccountingAudit, type SearchArtifactRef, type SearchAttemptAccounting, type SearchCandidateDecidedEvent, type SearchCandidateLineage, type SearchCandidateRegisteredEvent, type SearchCandidateSlot, type SearchCandidateSlotClosedEvent, type SearchCandidateSurface, type SearchCompletedEvent, type SearchCostAccounting, type SearchFailureReason, type SearchLedger, type SearchLedgerAppendResult, SearchLedgerConflictError, type SearchLedgerEntry, SearchLedgerError, type SearchLedgerEvent, type SearchLedgerHash, SearchLedgerIntegrityError, type SearchLedgerReplay, type SearchLedgerTrustedHeadMode, type SearchModelIdentity, type SearchOperationKind, type SearchOperationRecordedEvent, type SearchPlan, type SearchPlannedEvent, type SearchPlannedOperation, type SearchPlannedTask, type SearchSourceRef, type SearchSurfaceEffect, type SearchSurfaceEvidence, type SearchSurfaceKind, type SearchTaskAttemptedEvent, type SearchTaskOutcome, type SearchTokenAccounting, SegmentedProfileMatrixError, type SequentialDecideFn, type SequentialDecideOptions, type SequentialDecision, type SequentialObservation, type SequentialPairedGate, type SequentialPairedGateOptions, type SessionScript, type SingleRunLock, type SingleRunLockOptions, type SkillOptOptimizationMethodConfig, type SkillOptRunnerCommand, type SkillOptTrainerConfig, type SurfaceProposer, type TraceAnalystArtifact, type TraceAnalystScenario, type TraceSpan, type TransientFailureOptions, type UngroundedLiteralReport, type UserStory, type UserStoryVerdict, type Worktree, type WorktreeAdapter, WorktreeAdapterError, acquireSingleRunLock, analyzeCrossSurfaceInteractions, assertCampaignDesign, assertCampaignSplitIdentity, assertCodeSurfaceIdentity, assertComponentSurface, autoevalsScorerJudge, buildEvidenceVector, buildLoopProvenanceRecord, buildTraceAnalystSurfaceDispatch, campaignBreakdown, campaignMeanComposite, campaignMeasurementDigest, campaignScenarioIdentity, campaignSplitDigest, campaignSplitDigestFromIdentities, canonicalDigest, classifyUngroundedLiterals, codeSurfaceIdentityMaterial, combineComparisonCosts, compareOptimizationMethods, compareRankKeys, componentSurfaceIdentityMaterial, composeGate, costFromLedgerSummary, createProfileMatrixPlan, createReferenceEquivalenceJudge, createRunCostLedger, decodeExternalTextCandidate, defaultProductionGate, detectScale, dimensionRegressions, discoverEvalFixtures, emitLoopProvenance, externalTextOptimizationMethod, finalizeProfileMatrix, fsCampaignStorage, gepaOptimizationMethod, gitWorktreeAdapter, heldOutGate, heldoutSignificance, inMemoryCampaignStorage, isProposedCandidate, isTransientTransportFailure, labelTrustRank, llmJudge, loadEvalFixture, loadEvalFixtureScenarios, loopProvenanceArgsFromResult, loopProvenanceSpans, makePlaybackDispatch, makeProposalFinding, neutralizationGate, neutralizeText, openAutoPr, openSearchLedger, optimizationTokenUsageFromSummary, pairHoldout, paretoPolicy, paretoSignificanceGate, phoenixEvaluatorJudge, planCampaignRun, planEvalFixtureRun, powerPreflight, provenanceRecordPath, provenanceSpansPath, readExternalOptimizerObservationArtifact, readGepaCandidatePopulationArtifact, renderScoreboardMarkdown, renderSurfaceDiff, resolveExternalOptimizerCallbackLimits, resolveExternalOptimizerProcessLimits, resolveRunDir, resolveWorktreePath, rolloutArgumentDiff, runCampaign, runEval, runImprovementLoop, runOptimization, runProfileMatrix, runProfileMatrixSegment, scoreDiscrimination, scoreUserStory, scoreboardSummary, selectDiscriminative, sequentialDecide, sequentialPairedGate, skillOptOptimizationMethod, surfaceContentHash, surfaceHash, tangleTracesRoot, traceAnalystQualityJudge, userStoryScoreboard, validateSearchLedgerEvent, verifyCodeSurface, verifyLoopProvenanceRecord };
|
|
9
|
+
import { a as ObjectiveSource, c as PromotionPolicy, d as paretoSignificanceGate, i as EvidenceVector, l as buildEvidenceVector, n as AxisVerdict, o as ParetoSignificanceGateOptions, r as BuildEvidenceVectorOptions, s as PromotionObjective, t as AxisEvidence, u as paretoPolicy } from "../promotion-policy-u3wj6w3U.js";
|
|
10
|
+
import { a as detectScale, c as pairHoldout, d as SequentialDecision, f as SequentialObservation, g as sequentialPairedGate, h as sequentialDecide, i as PairedHoldout, l as SequentialDecideFn, m as SequentialPairedGateOptions, n as HeldoutSignificance, o as dimensionRegressions, p as SequentialPairedGate, r as HeldoutSignificanceOptions, s as heldoutSignificance, t as DimensionRegression, u as SequentialDecideOptions } from "../statistical-heldout-CBGPSZ5X.js";
|
|
11
|
+
export { type AnalyzeCrossSurfaceInteractionsInput, type AutoevalsScoreLike, type AutoevalsScorerLike, type AxisEvidence, type AxisVerdict, type BuildEvidenceVectorOptions, type BuildLoopProvenanceArgs, type BuildTraceAnalystSurfaceDispatchOptions, type CacheIssueReason, type CacheRead, type CampaignAggregates, type CampaignArtifactWriter, type CampaignBreakdown, type CampaignCellFailureReceipt, type CampaignCellResult, type CampaignCostMeter, type CampaignResult, type CampaignRunPlan, type CampaignRunPlanCell, type CampaignScenarioIdentity, type CampaignStorage, type CampaignTokenUsage, type CampaignTraceWriter, type CodeSurface, type CodeSurfaceVerification, type CompareOptimizationMethodsOptions, type ComparisonCost, type ComponentSurface, type CostLedgerHandle, type CreateProfileMatrixPlanOptions, type CrossSurfaceAdditionDecision, type CrossSurfaceAdditionRejectionReason, type CrossSurfaceAttemptCompleteness, type CrossSurfaceBestSingleSelection, type CrossSurfaceBootstrapPolicy, type CrossSurfaceCandidate, type CrossSurfaceCandidateComparison, type CrossSurfaceCandidateEvidence, type CrossSurfaceCandidateOutcome, type CrossSurfaceCandidateSummary, type CrossSurfaceComponent, type CrossSurfaceComponentEvidence, type CrossSurfaceCompositionStep, type CrossSurfaceDistribution, type CrossSurfaceEligibility, type CrossSurfaceEvidenceBreakdown, type CrossSurfaceIneligibilityReason, type CrossSurfaceInteractionAwareSelection, type CrossSurfaceInteractionEffect, type CrossSurfaceInteractionPath, type CrossSurfaceInteractionReport, type CrossSurfaceInteractionTask, type CrossSurfaceNaiveStackSelection, type CrossSurfacePairCompatibility, type CrossSurfacePairEvidence, type CrossSurfacePairIncompatibilityReason, type CrossSurfacePairwiseEntry, type CrossSurfaceRankedSingle, type CrossSurfaceRelativeCost, type CrossSurfaceSelectionPolicy, type CrossSurfaceSelections, type CrossSurfaceTaskRow, DEFAULT_EXTERNAL_OPTIMIZER_CALLBACK_LIMITS, DEFAULT_EXTERNAL_OPTIMIZER_PROCESS_LIMITS, type DefaultProductionGateCheck, type DefaultProductionGateOptions, type DefaultProductionRewardHackingOptions, type DimensionRegression, type DiscriminationScore, type DispatchContext, type DispatchFn, type EmitLoopProvenanceArgs, type EmitLoopProvenanceResult, type EvalFixture, type EvalFixtureFile, type EvalFixtureLoadOptions, type EvalFixtureRunPlan, type EvalFixtureScenario, type EvalFixtureValidationMode, type EvidenceVector, type ExternalOptimizationExample, type ExternalOptimizerCallbackLimits, type ExternalOptimizerChatRequest, type ExternalOptimizerEndpointFormat, type ExternalOptimizerEvaluationObservation, type ExternalOptimizerEvaluationRefusalReason, type ExternalOptimizerExecutionSummary, type ExternalOptimizerModelBudget, type ExternalOptimizerModelCall, type ExternalOptimizerModelCallRequest, type ExternalOptimizerModelCallResult, type ExternalOptimizerModelExecutionObservation, type ExternalOptimizerObservationArtifact, type ExternalOptimizerObservationSummary, type ExternalOptimizerProcessLimits, type ExternalOptimizerRunnerCommand, type ExternalOptimizerSubmittedCandidate, type ExternalTextCandidate, type ExternalTextEvaluationResponse, type ExternalTextOptimizationMethodConfig, type ExternalTextOptimizerContext, type ExternalTextOptimizerResult, FileSearchLedger, type FinalizeProfileMatrixOptions, type FinalizedProfileMatrixResult, FsLabeledScenarioStore, type FsLabeledScenarioStoreOptions, type Gate, type GateCheckStatus, type GateContext, type GateContribution, type GateDecision, type GateResult, type GenerationCandidate, type GenerationRecord, type GepaAdaptiveEngineRun, type GepaCandidatePopulationArtifact, type GepaCandidatePopulationCandidate, type GepaCandidatePopulationSummary, type GepaCandidateSelectionScore, type GepaEngineOptions, type GepaEngineRun, type GepaOptimizationMethodConfig, type GepaOptimizationRecipe, type GepaRunnerCommand, type GitWorktreeAdapterOptions, type HeldOutGateOptions, type HeldoutSignificance, type HeldoutSignificanceOptions, type JudgeAggregate, type JudgeConfig, type JudgeDimension, type JudgeScore, type LabelTrust, type LabeledScenarioRecord, type LabeledScenarioSampleArgs, type LabeledScenarioSource, type LabeledScenarioStore, LabeledScenarioStoreError, type LabeledScenarioWrite, type LlmJudgeDimension, type LlmJudgeOptions, type LoadEvalFixtureScenariosOptions, type LoopProvenanceArgsFromResult, type LoopProvenanceBackend, type LoopProvenanceCandidate, type LoopProvenanceEvidence, type LoopProvenanceOptimizationMethod, type LoopProvenanceRecord, type MutableSurface, type NeutralizationGateOptions, type ObjectiveSource, type OpenAICompatibleOptimizerModel, type OpenAutoPrOptions, type OpenAutoPrResult, type OpenSearchLedgerOptions, type OptimizationMethod, type OptimizationMethodComparison, type OptimizationMethodInput, type OptimizationMethodPairwise, type OptimizationMethodProvenance, type OptimizationMethodResult, type OptimizationMethodRunOptions, type OptimizationMethodScore, type OptimizationPackageSource, type OptimizationTokenUsage, type OptimizerConfig, type OptimizerModelBudget, type PairedHoldout, type ParetoParent, type ParetoSignificanceGateOptions, type PendingCostCallView, type PhoenixEvaluationResultLike, type PhoenixEvaluatorLike, type PlanCampaignRunOptions, type PlanEvalFixtureRunOptions, type PlaybackContext, type PlaybackDriver, type PlaybackStep, type PowerPreflight, type PowerPreflightOptions, type PremeasuredOptimizationBaseline, type ProfileDispatchFn, type ProfileMatrixCoverage, ProfileMatrixError, type ProfileMatrixPlan, type ProfileMatrixRow, type ProfileMatrixSegmentResult, type ProfileSummary, type PromotionObjective, type PromotionPolicy, type ProposalFinding, type ProposalFindingOrigin, type ProposalTrackContext, type ProposeContext, type ProposedCandidate, type RedactionStatus, type ReferenceEquivalenceJudgeOptions, type ReferenceEquivalenceScenario, type RolloutArgumentDiff, type RolloutArgumentDiffOptions, type RolloutCall, type RunCampaignOptions, type RunEvalOptions, type RunImprovementLoopOptions, type RunImprovementLoopResult, type RunOptimizationOptions, type RunOptimizationResult, type RunProfileMatrixOptions, type RunProfileMatrixResult, type RunProfileMatrixSegmentOptions, SEARCH_LEDGER_SCHEMA, type Scenario, type ScenarioAggregate, type ScenarioRollup, type ScenarioSignal, type ScoreboardRenderOptions, type ScoreboardRow, type ScoreboardSummary, type ScoredRollout, type ScoredSurfaceOutcome, type SearchAccountingAudit, type SearchArtifactRef, type SearchAttemptAccounting, type SearchCandidateDecidedEvent, type SearchCandidateLineage, type SearchCandidateRegisteredEvent, type SearchCandidateSlot, type SearchCandidateSlotClosedEvent, type SearchCandidateSurface, type SearchCompletedEvent, type SearchCostAccounting, type SearchFailureReason, type SearchLedger, type SearchLedgerAppendResult, SearchLedgerConflictError, type SearchLedgerEntry, SearchLedgerError, type SearchLedgerEvent, type SearchLedgerHash, SearchLedgerIntegrityError, type SearchLedgerReplay, type SearchLedgerTrustedHeadMode, type SearchModelIdentity, type SearchOperationKind, type SearchOperationRecordedEvent, type SearchPlan, type SearchPlannedEvent, type SearchPlannedOperation, type SearchPlannedTask, type SearchSourceRef, type SearchSurfaceEffect, type SearchSurfaceEvidence, type SearchSurfaceKind, type SearchTaskAttemptedEvent, type SearchTaskOutcome, type SearchTokenAccounting, SegmentedProfileMatrixError, type SequentialDecideFn, type SequentialDecideOptions, type SequentialDecision, type SequentialObservation, type SequentialPairedGate, type SequentialPairedGateOptions, type SessionScript, type SingleRunLock, type SingleRunLockOptions, type SkillOptOptimizationMethodConfig, type SkillOptRunnerCommand, type SkillOptTrainerConfig, type SurfaceProposer, type TraceAnalystArtifact, type TraceAnalystScenario, type TraceSpan, type TransientFailureOptions, type UngroundedLiteralReport, type UserStory, type UserStoryVerdict, type Worktree, type WorktreeAdapter, WorktreeAdapterError, acquireSingleRunLock, analyzeCrossSurfaceInteractions, assertCampaignDesign, assertCampaignSplitIdentity, assertCodeSurfaceIdentity, assertComponentSurface, autoevalsScorerJudge, buildEvidenceVector, buildLoopProvenanceRecord, buildTraceAnalystSurfaceDispatch, campaignBreakdown, campaignMeanComposite, campaignMeasurementDigest, campaignScenarioIdentity, campaignSplitDigest, campaignSplitDigestFromIdentities, canonicalDigest, cellCachePath, classifyUngroundedLiterals, codeSurfaceIdentityMaterial, combineComparisonCosts, compareOptimizationMethods, compareRankKeys, componentSurfaceIdentityMaterial, composeGate, costFromLedgerSummary, createProfileMatrixPlan, createReferenceEquivalenceJudge, createRunCostLedger, decodeExternalTextCandidate, defaultProductionGate, detectScale, dimensionRegressions, discoverEvalFixtures, emitLoopProvenance, externalTextOptimizationMethod, finalizeProfileMatrix, fsCampaignStorage, gepaOptimizationMethod, gitWorktreeAdapter, heldOutGate, heldoutSignificance, inMemoryCampaignStorage, isProposedCandidate, isTransientTransportFailure, labelTrustRank, llmJudge, loadEvalFixture, loadEvalFixtureScenarios, loopProvenanceArgsFromResult, loopProvenanceSpans, makePlaybackDispatch, makeProposalFinding, neutralizationGate, neutralizeText, openAutoPr, openSearchLedger, optimizationTokenUsageFromSummary, pairHoldout, paretoPolicy, paretoSignificanceGate, phoenixEvaluatorJudge, planCampaignRun, planEvalFixtureRun, powerPreflight, provenanceRecordPath, provenanceSpansPath, readCachedCell, readExternalOptimizerObservationArtifact, readGepaCandidatePopulationArtifact, renderScoreboardMarkdown, renderSurfaceDiff, resolveExternalOptimizerCallbackLimits, resolveExternalOptimizerProcessLimits, resolveRunDir, resolveWorktreePath, rolloutArgumentDiff, runCampaign, runEval, runImprovementLoop, runOptimization, runProfileMatrix, runProfileMatrixSegment, scoreDiscrimination, scoreUserStory, scoreboardSummary, selectDiscriminative, sequentialDecide, sequentialPairedGate, skillOptOptimizationMethod, surfaceContentHash, surfaceHash, tangleTracesRoot, traceAnalystQualityJudge, userStoryScoreboard, validateSearchLedgerEvent, verifyCodeSurface, verifyLoopProvenanceRecord };
|
package/dist/campaign/index.js
CHANGED
|
@@ -1,10 +1,10 @@
|
|
|
1
|
-
import { $ as assertCampaignDesign, A as surfaceHash, C as optimizationTokenUsageFromSummary, D as componentSurfaceIdentityMaterial, E as codeSurfaceIdentityMaterial, G as runEval, J as resolveRunDir, K as runCampaign, M as campaignMeanComposite, N as compareRankKeys, O as renderSurfaceDiff, R as readGepaCandidatePopulationArtifact, S as costFromLedgerSummary, T as assertComponentSurface, Y as tangleTracesRoot, a as canonicalDigest, b as combineComparisonCosts, c as loopProvenanceSpans, d as verifyLoopProvenanceRecord, et as assertCampaignSplitIdentity, f as runImprovementLoop, h as labelTrustRank, i as campaignMeasurementDigest, j as campaignBreakdown, k as surfaceContentHash, l as provenanceRecordPath, m as isProposedCandidate, nt as campaignSplitDigest, o as emitLoopProvenance, p as runOptimization, q as planCampaignRun, r as buildLoopProvenanceRecord, rt as campaignSplitDigestFromIdentities, s as loopProvenanceArgsFromResult, t as llmJudge, tt as campaignScenarioIdentity, u as provenanceSpansPath, v as openAutoPr, w as assertCodeSurfaceIdentity, x as compareOptimizationMethods, z as defaultProductionGate } from "../llm-judge-
|
|
1
|
+
import { $ as assertCampaignDesign, A as surfaceHash, C as optimizationTokenUsageFromSummary, D as componentSurfaceIdentityMaterial, E as codeSurfaceIdentityMaterial, G as runEval, J as resolveRunDir, K as runCampaign, M as campaignMeanComposite, N as compareRankKeys, O as renderSurfaceDiff, R as readGepaCandidatePopulationArtifact, S as costFromLedgerSummary, T as assertComponentSurface, Y as tangleTracesRoot, a as canonicalDigest, at as cellCachePath, b as combineComparisonCosts, c as loopProvenanceSpans, d as verifyLoopProvenanceRecord, et as assertCampaignSplitIdentity, f as runImprovementLoop, h as labelTrustRank, i as campaignMeasurementDigest, it as readCachedCell, j as campaignBreakdown, k as surfaceContentHash, l as provenanceRecordPath, m as isProposedCandidate, nt as campaignSplitDigest, o as emitLoopProvenance, p as runOptimization, q as planCampaignRun, r as buildLoopProvenanceRecord, rt as campaignSplitDigestFromIdentities, s as loopProvenanceArgsFromResult, t as llmJudge, tt as campaignScenarioIdentity, u as provenanceSpansPath, v as openAutoPr, w as assertCodeSurfaceIdentity, x as compareOptimizationMethods, z as defaultProductionGate } from "../llm-judge-DhiJqSjB.js";
|
|
2
2
|
import { E as SearchLedgerIntegrityError, S as inMemoryCampaignStorage, T as SearchLedgerError, _ as resolveExternalOptimizerCallbackLimits, b as createRunCostLedger, c as DEFAULT_EXTERNAL_OPTIMIZER_CALLBACK_LIMITS, l as DEFAULT_EXTERNAL_OPTIMIZER_PROCESS_LIMITS, v as resolveExternalOptimizerProcessLimits, w as SearchLedgerConflictError, x as fsCampaignStorage } from "../external-optimizer-subprocess-BhKYK0Jv.js";
|
|
3
3
|
import { a as heldoutSignificance, i as dimensionRegressions, o as pairHoldout, r as detectScale, t as powerPreflight } from "../power-preflight-DEw-uC7q.js";
|
|
4
4
|
import { r as makeProposalFinding } from "../types-BI4fT3HN.js";
|
|
5
5
|
import { n as acquireSingleRunLock } from "../external-optimizer-process-BRE56woM.js";
|
|
6
|
-
import { a as externalTextOptimizationMethod, i as composeGate, n as gepaOptimizationMethod, o as decodeExternalTextCandidate, r as heldOutGate, s as readExternalOptimizerObservationArtifact, t as skillOptOptimizationMethod, u as createReferenceEquivalenceJudge } from "../skillopt-optimization-method-
|
|
7
|
-
import { A as neutralizationGate, C as scoreboardSummary, D as LabeledScenarioStoreError, E as FsLabeledScenarioStore, F as analyzeCrossSurfaceInteractions, I as buildTraceAnalystSurfaceDispatch, L as traceAnalystQualityJudge, M as loadEvalFixture, N as loadEvalFixtureScenarios, O as classifyUngroundedLiterals, P as planEvalFixtureRun, S as scoreUserStory, T as neutralizeText, _ as runProfileMatrixSegment, a as autoevalsScorerJudge, b as makePlaybackDispatch, c as FileSearchLedger, d as validateSearchLedgerEvent, f as scoreDiscrimination, g as finalizeProfileMatrix, h as createProfileMatrixPlan, i as verifyCodeSurface, j as discoverEvalFixtures, k as rolloutArgumentDiff, l as SEARCH_LEDGER_SCHEMA, m as SegmentedProfileMatrixError, n as gitWorktreeAdapter, o as phoenixEvaluatorJudge, p as selectDiscriminative, r as resolveWorktreePath, s as isTransientTransportFailure, t as WorktreeAdapterError, u as openSearchLedger, v as ProfileMatrixError, w as userStoryScoreboard, x as renderScoreboardMarkdown, y as runProfileMatrix } from "../campaign-
|
|
6
|
+
import { a as externalTextOptimizationMethod, i as composeGate, n as gepaOptimizationMethod, o as decodeExternalTextCandidate, r as heldOutGate, s as readExternalOptimizerObservationArtifact, t as skillOptOptimizationMethod, u as createReferenceEquivalenceJudge } from "../skillopt-optimization-method-C6JSfYxT.js";
|
|
7
|
+
import { A as neutralizationGate, C as scoreboardSummary, D as LabeledScenarioStoreError, E as FsLabeledScenarioStore, F as analyzeCrossSurfaceInteractions, I as buildTraceAnalystSurfaceDispatch, L as traceAnalystQualityJudge, M as loadEvalFixture, N as loadEvalFixtureScenarios, O as classifyUngroundedLiterals, P as planEvalFixtureRun, S as scoreUserStory, T as neutralizeText, _ as runProfileMatrixSegment, a as autoevalsScorerJudge, b as makePlaybackDispatch, c as FileSearchLedger, d as validateSearchLedgerEvent, f as scoreDiscrimination, g as finalizeProfileMatrix, h as createProfileMatrixPlan, i as verifyCodeSurface, j as discoverEvalFixtures, k as rolloutArgumentDiff, l as SEARCH_LEDGER_SCHEMA, m as SegmentedProfileMatrixError, n as gitWorktreeAdapter, o as phoenixEvaluatorJudge, p as selectDiscriminative, r as resolveWorktreePath, s as isTransientTransportFailure, t as WorktreeAdapterError, u as openSearchLedger, v as ProfileMatrixError, w as userStoryScoreboard, x as renderScoreboardMarkdown, y as runProfileMatrix } from "../campaign-Dg2B4W6d.js";
|
|
8
8
|
import { n as paretoPolicy, r as paretoSignificanceGate, t as buildEvidenceVector } from "../promotion-policy-xzA40Evo.js";
|
|
9
9
|
import { n as sequentialPairedGate, t as sequentialDecide } from "../sequential-C458DXNf.js";
|
|
10
|
-
export { DEFAULT_EXTERNAL_OPTIMIZER_CALLBACK_LIMITS, DEFAULT_EXTERNAL_OPTIMIZER_PROCESS_LIMITS, FileSearchLedger, FsLabeledScenarioStore, LabeledScenarioStoreError, ProfileMatrixError, SEARCH_LEDGER_SCHEMA, SearchLedgerConflictError, SearchLedgerError, SearchLedgerIntegrityError, SegmentedProfileMatrixError, WorktreeAdapterError, acquireSingleRunLock, analyzeCrossSurfaceInteractions, assertCampaignDesign, assertCampaignSplitIdentity, assertCodeSurfaceIdentity, assertComponentSurface, autoevalsScorerJudge, buildEvidenceVector, buildLoopProvenanceRecord, buildTraceAnalystSurfaceDispatch, campaignBreakdown, campaignMeanComposite, campaignMeasurementDigest, campaignScenarioIdentity, campaignSplitDigest, campaignSplitDigestFromIdentities, canonicalDigest, classifyUngroundedLiterals, codeSurfaceIdentityMaterial, combineComparisonCosts, compareOptimizationMethods, compareRankKeys, componentSurfaceIdentityMaterial, composeGate, costFromLedgerSummary, createProfileMatrixPlan, createReferenceEquivalenceJudge, createRunCostLedger, decodeExternalTextCandidate, defaultProductionGate, detectScale, dimensionRegressions, discoverEvalFixtures, emitLoopProvenance, externalTextOptimizationMethod, finalizeProfileMatrix, fsCampaignStorage, gepaOptimizationMethod, gitWorktreeAdapter, heldOutGate, heldoutSignificance, inMemoryCampaignStorage, isProposedCandidate, isTransientTransportFailure, labelTrustRank, llmJudge, loadEvalFixture, loadEvalFixtureScenarios, loopProvenanceArgsFromResult, loopProvenanceSpans, makePlaybackDispatch, makeProposalFinding, neutralizationGate, neutralizeText, openAutoPr, openSearchLedger, optimizationTokenUsageFromSummary, pairHoldout, paretoPolicy, paretoSignificanceGate, phoenixEvaluatorJudge, planCampaignRun, planEvalFixtureRun, powerPreflight, provenanceRecordPath, provenanceSpansPath, readExternalOptimizerObservationArtifact, readGepaCandidatePopulationArtifact, renderScoreboardMarkdown, renderSurfaceDiff, resolveExternalOptimizerCallbackLimits, resolveExternalOptimizerProcessLimits, resolveRunDir, resolveWorktreePath, rolloutArgumentDiff, runCampaign, runEval, runImprovementLoop, runOptimization, runProfileMatrix, runProfileMatrixSegment, scoreDiscrimination, scoreUserStory, scoreboardSummary, selectDiscriminative, sequentialDecide, sequentialPairedGate, skillOptOptimizationMethod, surfaceContentHash, surfaceHash, tangleTracesRoot, traceAnalystQualityJudge, userStoryScoreboard, validateSearchLedgerEvent, verifyCodeSurface, verifyLoopProvenanceRecord };
|
|
10
|
+
export { DEFAULT_EXTERNAL_OPTIMIZER_CALLBACK_LIMITS, DEFAULT_EXTERNAL_OPTIMIZER_PROCESS_LIMITS, FileSearchLedger, FsLabeledScenarioStore, LabeledScenarioStoreError, ProfileMatrixError, SEARCH_LEDGER_SCHEMA, SearchLedgerConflictError, SearchLedgerError, SearchLedgerIntegrityError, SegmentedProfileMatrixError, WorktreeAdapterError, acquireSingleRunLock, analyzeCrossSurfaceInteractions, assertCampaignDesign, assertCampaignSplitIdentity, assertCodeSurfaceIdentity, assertComponentSurface, autoevalsScorerJudge, buildEvidenceVector, buildLoopProvenanceRecord, buildTraceAnalystSurfaceDispatch, campaignBreakdown, campaignMeanComposite, campaignMeasurementDigest, campaignScenarioIdentity, campaignSplitDigest, campaignSplitDigestFromIdentities, canonicalDigest, cellCachePath, classifyUngroundedLiterals, codeSurfaceIdentityMaterial, combineComparisonCosts, compareOptimizationMethods, compareRankKeys, componentSurfaceIdentityMaterial, composeGate, costFromLedgerSummary, createProfileMatrixPlan, createReferenceEquivalenceJudge, createRunCostLedger, decodeExternalTextCandidate, defaultProductionGate, detectScale, dimensionRegressions, discoverEvalFixtures, emitLoopProvenance, externalTextOptimizationMethod, finalizeProfileMatrix, fsCampaignStorage, gepaOptimizationMethod, gitWorktreeAdapter, heldOutGate, heldoutSignificance, inMemoryCampaignStorage, isProposedCandidate, isTransientTransportFailure, labelTrustRank, llmJudge, loadEvalFixture, loadEvalFixtureScenarios, loopProvenanceArgsFromResult, loopProvenanceSpans, makePlaybackDispatch, makeProposalFinding, neutralizationGate, neutralizeText, openAutoPr, openSearchLedger, optimizationTokenUsageFromSummary, pairHoldout, paretoPolicy, paretoSignificanceGate, phoenixEvaluatorJudge, planCampaignRun, planEvalFixtureRun, powerPreflight, provenanceRecordPath, provenanceSpansPath, readCachedCell, readExternalOptimizerObservationArtifact, readGepaCandidatePopulationArtifact, renderScoreboardMarkdown, renderSurfaceDiff, resolveExternalOptimizerCallbackLimits, resolveExternalOptimizerProcessLimits, resolveRunDir, resolveWorktreePath, rolloutArgumentDiff, runCampaign, runEval, runImprovementLoop, runOptimization, runProfileMatrix, runProfileMatrixSegment, scoreDiscrimination, scoreUserStory, scoreboardSummary, selectDiscriminative, sequentialDecide, sequentialPairedGate, skillOptOptimizationMethod, surfaceContentHash, surfaceHash, tangleTracesRoot, traceAnalystQualityJudge, userStoryScoreboard, validateSearchLedgerEvent, verifyCodeSurface, verifyLoopProvenanceRecord };
|
|
@@ -1,12 +1,12 @@
|
|
|
1
1
|
import { s as ValidationError, t as AgentEvalError } from "./errors-Dngq5h35.js";
|
|
2
|
-
import { c as agentProfileId, d as harnessAxisOf, i as verifyCompletion, l as agentProfileModelId, s as agentProfileHash, t as extractProducedState } from "./produced-state-
|
|
2
|
+
import { c as agentProfileId, d as harnessAxisOf, i as verifyCompletion, l as agentProfileModelId, s as agentProfileHash, t as extractProducedState } from "./produced-state-BNyyud4g.js";
|
|
3
3
|
import { buildAgentProfileCell } from "./profile-cell.js";
|
|
4
4
|
import { t as comparePairedArms } from "./paired-arms-D-XRF_fy.js";
|
|
5
5
|
import { r as pairedBootstrap } from "./paired-tests-BHIhYVdu.js";
|
|
6
|
-
import {
|
|
6
|
+
import { c as validateRunRecord, i as modelHasSnapshot, n as UNKNOWN_MODEL, s as runTaskScore } from "./run-record-D2lDdSAz.js";
|
|
7
7
|
import { n as contentHash, t as canonicalJson } from "./verdict-cache-mZf5FEiY.js";
|
|
8
|
-
import { $ as assertCampaignDesign, J as resolveRunDir, K as runCampaign, Q as summarizeBackendIntegrity, Z as assertRealBackend, at as
|
|
9
|
-
import { a as campaignCellCostProvenance, l as campaignCellToRunRecord, u as projectCampaignCellQuality } from "./reward-hacking-
|
|
8
|
+
import { $ as assertCampaignDesign, J as resolveRunDir, K as runCampaign, Q as summarizeBackendIntegrity, Z as assertRealBackend, at as cellCachePath, h as labelTrustRank, k as surfaceContentHash, ot as computeManifestHash, q as planCampaignRun, tt as campaignScenarioIdentity, w as assertCodeSurfaceIdentity } from "./llm-judge-DhiJqSjB.js";
|
|
9
|
+
import { a as campaignCellCostProvenance, l as campaignCellToRunRecord, u as projectCampaignCellQuality } from "./reward-hacking-DFo2FU5J.js";
|
|
10
10
|
import { t as CostAccountingIncompleteError } from "./cost-ledger-BSe92yAV.js";
|
|
11
11
|
import { t as FileLedgerJournal, u as mapConcurrent } from "./ledger-core-BmZt19oQ.js";
|
|
12
12
|
import { r as canonicalString } from "./canonical-D-XsTQ6_.js";
|
|
@@ -14,7 +14,7 @@ import { C as SEARCH_LEDGER_FILE_CONTEXT, E as SearchLedgerIntegrityError, T as
|
|
|
14
14
|
import { o as pairHoldout } from "./power-preflight-DEw-uC7q.js";
|
|
15
15
|
import { a as scoreAnalystFindings } from "./benchmark-BhT16ep9.js";
|
|
16
16
|
import "./external-optimizer-process-BRE56woM.js";
|
|
17
|
-
import "./skillopt-optimization-method-
|
|
17
|
+
import "./skillopt-optimization-method-C6JSfYxT.js";
|
|
18
18
|
import { createHash } from "node:crypto";
|
|
19
19
|
import { closeSync, constants, existsSync, fstatSync, lstatSync, mkdirSync, mkdtempSync, openSync, readFileSync, readSync, readdirSync, readlinkSync, realpathSync, rmSync, statSync, writeFileSync } from "node:fs";
|
|
20
20
|
import { z } from "zod";
|
|
@@ -1545,6 +1545,7 @@ function receiptModels(cell) {
|
|
|
1545
1545
|
/** Resolve and validate every paid-call model used by one cell. */
|
|
1546
1546
|
function recordModel(cell, profileId, declaredModel) {
|
|
1547
1547
|
const reported = receiptModels(cell);
|
|
1548
|
+
if (reported.length === 0 && cell.error !== void 0) return UNKNOWN_MODEL;
|
|
1548
1549
|
if (modelHasSnapshot(declaredModel)) {
|
|
1549
1550
|
for (const model of reported) if (model !== declaredModel) throw new ProfileMatrixError(`profile '${profileId}' paid-call model '${model}' for cell '${cell.cellId}' does not match its declared exact model '${declaredModel}'`);
|
|
1550
1551
|
return declaredModel;
|
|
@@ -1565,7 +1566,7 @@ function buildRunRecord(args) {
|
|
|
1565
1566
|
const { cell, profile, profileHash, configHash, experimentId, splitTag, commitSha, matrixId } = args;
|
|
1566
1567
|
const profileId = agentProfileId(profile);
|
|
1567
1568
|
const model = recordModel(cell, profileId, agentProfileModelId(profile));
|
|
1568
|
-
const record = campaignCellToRunRecord(cell, {
|
|
1569
|
+
const record = campaignCellToRunRecord(durableCellForRecord(cell), {
|
|
1569
1570
|
runId: `${matrixId}:${profileId}:${cell.cellId}`,
|
|
1570
1571
|
experimentId,
|
|
1571
1572
|
candidateId: profileId,
|
|
@@ -1586,6 +1587,28 @@ function buildRunRecord(args) {
|
|
|
1586
1587
|
return record;
|
|
1587
1588
|
}
|
|
1588
1589
|
/**
|
|
1590
|
+
* Keep failed rows truthful at the RunRecord boundary.
|
|
1591
|
+
*
|
|
1592
|
+
* A cell with no settled call currently has a zero subtotal and an observed
|
|
1593
|
+
* zero from the ledger. That is not measured provider usage or cost, so mark
|
|
1594
|
+
* usage incomplete and cost uncaptured while retaining the known subtotal.
|
|
1595
|
+
*/
|
|
1596
|
+
function durableCellForRecord(cell) {
|
|
1597
|
+
if (cell.error === void 0) return cell;
|
|
1598
|
+
const hasSettledAgentCalls = receiptModels(cell).length > 0;
|
|
1599
|
+
return {
|
|
1600
|
+
...cell,
|
|
1601
|
+
tokenUsage: cell.tokenUsage.tokensKnown === false || hasSettledAgentCalls ? cell.tokenUsage : {
|
|
1602
|
+
...cell.tokenUsage,
|
|
1603
|
+
tokensKnown: false
|
|
1604
|
+
},
|
|
1605
|
+
costProvenance: !hasSettledAgentCalls && cell.costProvenance.kind === "observed" && cell.costUsd === 0 ? {
|
|
1606
|
+
kind: "uncaptured",
|
|
1607
|
+
usd: null
|
|
1608
|
+
} : cell.costProvenance
|
|
1609
|
+
};
|
|
1610
|
+
}
|
|
1611
|
+
/**
|
|
1589
1612
|
* Profile × scenario matrix runner: fan N agent profiles across M scenarios, project each cell to a validated `RunRecord` with real token usage, and enforce the backend-integrity guard before returning.
|
|
1590
1613
|
*/
|
|
1591
1614
|
async function runProfileMatrix(opts) {
|
|
@@ -1695,13 +1718,14 @@ async function runProfileMatrix(opts) {
|
|
|
1695
1718
|
kind: "agent-interface-profile",
|
|
1696
1719
|
hash: profileHash
|
|
1697
1720
|
},
|
|
1698
|
-
model: cellModel,
|
|
1721
|
+
...cellModel && cellModel !== "unknown" ? { model: cellModel } : {},
|
|
1699
1722
|
...axis ? { harness: { id: axis.harness } } : {}
|
|
1700
1723
|
});
|
|
1701
1724
|
const sharedCellIdentity = modelHasSnapshot(declaredModel) ? await buildCellIdentity(declaredModel) : void 0;
|
|
1702
1725
|
const profileRecords = [];
|
|
1703
1726
|
for (const cell of campaign.cells) {
|
|
1704
|
-
const
|
|
1727
|
+
const cellModel = recordModel(cell, profileId, declaredModel);
|
|
1728
|
+
const agentProfileCell = cellModel === "unknown" ? await buildCellIdentity() : sharedCellIdentity ?? await buildCellIdentity(cellModel);
|
|
1705
1729
|
const record = buildRunRecord({
|
|
1706
1730
|
cell,
|
|
1707
1731
|
profile,
|
|
@@ -1718,7 +1742,7 @@ async function runProfileMatrix(opts) {
|
|
|
1718
1742
|
if (validate) validateRunRecord(record);
|
|
1719
1743
|
profileRecords.push(record);
|
|
1720
1744
|
}
|
|
1721
|
-
const recordedModels = [...new Set(profileRecords.map((record) => record.model))];
|
|
1745
|
+
const recordedModels = [...new Set(profileRecords.map((record) => record.model).filter((model) => model !== UNKNOWN_MODEL))];
|
|
1722
1746
|
if (recordedModels.length > 1) throw new ProfileMatrixError(`profile '${profileId}' resolved to multiple model snapshots: ${recordedModels.join(", ")}`);
|
|
1723
1747
|
const costProvenance = campaign.aggregates.cost.costProvenance;
|
|
1724
1748
|
const totalCostUsd = costProvenance.kind === "uncaptured" ? null : costProvenance.usd;
|
|
@@ -1729,7 +1753,7 @@ async function runProfileMatrix(opts) {
|
|
|
1729
1753
|
summary: {
|
|
1730
1754
|
profileId,
|
|
1731
1755
|
profileHash,
|
|
1732
|
-
model: recordedModels[0] ??
|
|
1756
|
+
model: recordedModels[0] ?? "unknown",
|
|
1733
1757
|
records: profileRecords.length,
|
|
1734
1758
|
meanComposite: meanOrNull(profileRecords.map(scoreOf).filter((score) => score !== void 0)),
|
|
1735
1759
|
totalCostUsd,
|
|
@@ -3658,4 +3682,4 @@ function resolveWorktreePath(surface, worktreeDir) {
|
|
|
3658
3682
|
//#endregion
|
|
3659
3683
|
export { neutralizationGate as A, scoreboardSummary as C, LabeledScenarioStoreError as D, FsLabeledScenarioStore as E, analyzeCrossSurfaceInteractions as F, buildTraceAnalystSurfaceDispatch as I, traceAnalystQualityJudge as L, loadEvalFixture as M, loadEvalFixtureScenarios as N, classifyUngroundedLiterals as O, planEvalFixtureRun as P, scoreUserStory as S, neutralizeText as T, runProfileMatrixSegment as _, autoevalsScorerJudge as a, makePlaybackDispatch as b, FileSearchLedger as c, validateSearchLedgerEvent as d, scoreDiscrimination as f, finalizeProfileMatrix as g, createProfileMatrixPlan as h, verifyCodeSurface as i, discoverEvalFixtures as j, rolloutArgumentDiff as k, SEARCH_LEDGER_SCHEMA as l, SegmentedProfileMatrixError as m, gitWorktreeAdapter as n, phoenixEvaluatorJudge as o, selectDiscriminative as p, resolveWorktreePath as r, isTransientTransportFailure as s, WorktreeAdapterError as t, openSearchLedger as u, ProfileMatrixError as v, userStoryScoreboard as w, renderScoreboardMarkdown as x, runProfileMatrix as y };
|
|
3660
3684
|
|
|
3661
|
-
//# sourceMappingURL=campaign-
|
|
3685
|
+
//# sourceMappingURL=campaign-Dg2B4W6d.js.map
|