@tangle-network/agent-eval 0.144.10 → 0.144.11
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +11 -0
- package/README.md +3 -0
- package/dist/analyst/index.d.ts +1 -1
- package/dist/analyst/index.js +1 -1
- package/dist/{benchmark-command-D3fEnKdf.js → benchmark-command-BteMFN62.js} +2 -2
- package/dist/{benchmark-command-D3fEnKdf.js.map → benchmark-command-BteMFN62.js.map} +1 -1
- package/dist/benchmarks/index.d.ts +1 -1
- package/dist/benchmarks/index.js +1 -1
- package/dist/{benchmarks-BzqyTvAr.js → benchmarks-Dzs8CKb1.js} +3 -3
- package/dist/{benchmarks-BzqyTvAr.js.map → benchmarks-Dzs8CKb1.js.map} +1 -1
- package/dist/campaign/index.d.ts +3 -3
- package/dist/campaign/index.js +3 -3
- package/dist/{campaign-ccOrVCuR.js → campaign-C2TTzQII.js} +2 -2
- package/dist/{campaign-ccOrVCuR.js.map → campaign-C2TTzQII.js.map} +1 -1
- package/dist/cli.js +1 -1
- package/dist/contract/index.d.ts +1 -1
- package/dist/contract/index.js +1 -1
- package/dist/{index-DvvvSho7.d.ts → index-DPPGNJ_R.d.ts} +2 -2
- package/dist/{index-DvvvSho7.d.ts.map → index-DPPGNJ_R.d.ts.map} +1 -1
- package/dist/{index-CwYZNzMD2.d.ts → index-YE4KdKbO2.d.ts} +3 -3
- package/dist/{index-CwYZNzMD2.d.ts.map → index-YE4KdKbO2.d.ts.map} +1 -1
- package/dist/index.d.ts +3 -3
- package/dist/index.js +3 -3
- package/dist/openapi.json +1 -1
- package/dist/{skillopt-optimization-method-D6Q4dLbh.js → skillopt-optimization-method-CQdVeM8k.js} +219 -7
- package/dist/skillopt-optimization-method-CQdVeM8k.js.map +1 -0
- package/dist/{skillopt-optimization-method-9J8dJbrM.d.ts → skillopt-optimization-method-USDKhxSA.d.ts} +54 -2
- package/dist/skillopt-optimization-method-USDKhxSA.d.ts.map +1 -0
- package/docs/campaign-proposers.md +4 -0
- package/package.json +1 -1
- package/dist/skillopt-optimization-method-9J8dJbrM.d.ts.map +0 -1
- package/dist/skillopt-optimization-method-D6Q4dLbh.js.map +0 -1
package/dist/cli.js
CHANGED
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
#!/usr/bin/env node
|
|
2
2
|
import { o as runRolloutReleaseCli } from "./hf-dataset-XggBupCr.js";
|
|
3
|
-
import { n as runAnalystBenchmarkCommand } from "./benchmark-command-
|
|
3
|
+
import { n as runAnalystBenchmarkCommand } from "./benchmark-command-BteMFN62.js";
|
|
4
4
|
import { a as runRpcBatch, o as runRpcOnce, p as handleVersion, r as startServerAsync, s as buildOpenApi } from "./server-iu0ede49.js";
|
|
5
5
|
import { writeFileSync } from "node:fs";
|
|
6
6
|
//#region src/cli-config.ts
|
package/dist/contract/index.d.ts
CHANGED
|
@@ -6,7 +6,7 @@ import { n as buildDefaultAnalystRegistry, t as DefaultAnalystRegistryOptions }
|
|
|
6
6
|
import { C as JudgeDimension, H as SurfaceProposer, M as OptimizerConfig, R as Scenario, S as JudgeConfig, V as SessionScript, _ as GateDecision, a as CampaignResult, b as GenerationRecord, c as CampaignTraceWriter, d as DispatchContext, f as DispatchFn, g as GateContribution, h as GateContext, i as CampaignCostMeter, j as MutableSurface, k as LabeledScenarioStore, l as CodeSurface, m as GateCheckStatus, n as CampaignArtifactWriter, p as Gate, r as CampaignCellResult, t as CampaignAggregates, v as GateResult, w as JudgeScore, y as GenerationCandidate } from "../types-BnjdJ70P.js";
|
|
7
7
|
import { a as FailureClusterInsight, c as JudgeInsight, d as Recommendation, f as ReleaseSummary, i as FailureClassTally, l as LiftInsight, m as TokenUsageInsight, n as ExecutionErrorOutcomeCell, o as InsightReport, p as ScalarDistribution, r as ExecutionInsight, s as InterRaterInsight, t as CostProvenanceSummary, u as OutcomeCorrelationInsight } from "../insight-report-CgX_s0Ez.js";
|
|
8
8
|
import { D as analyzeRuns, E as SummarizeExecutionOptions, O as summarizeExecution, T as ExecutionReport, w as AnalyzeRunsOptions } from "../engine-3hL-XqwJ.js";
|
|
9
|
-
import { $t as
|
|
9
|
+
import { $t as fsCampaignStorage, A as RunEvalOptions, B as OptimizerModelBudget, Bt as runCampaign, C as RunImprovementLoopOptions, D as RunOptimizationOptions, E as PremeasuredOptimizationBaseline, F as GepaOptimizationMethodConfig, G as DefaultProductionRewardHackingOptions, H as heldOutGate, I as GepaOptimizationRecipe, K as defaultProductionGate, L as GepaRunnerCommand, M as GepaAdaptiveEngineRun, N as GepaEngineOptions, Ot as OptimizationPackageSource, P as GepaEngineRun, Pt as CampaignCellFailureReceipt, R as gepaOptimizationMethod, Rt as RunCampaignOptions, St as OptimizationMethodInput, T as runImprovementLoop, Tt as OptimizationMethodResult, U as DefaultProductionGateCheck, V as HeldOutGateOptions, W as DefaultProductionGateOptions, Zt as CampaignStorage, _n as LlmJudgeDimension, bt as OptimizationMethod, dn as ReferenceEquivalenceJudgeInput, dt as externalTextOptimizationMethod, en as inMemoryCampaignStorage, fn as ReferenceEquivalenceJudgeOptions, ft as ExternalTextOptimizationMethodConfig, gn as runReferenceEquivalenceJudge, gt as ExternalTextEvaluationResponse, hn as createReferenceEquivalenceJudge, ht as ExternalOptimizationExample, i as skillOptOptimizationMethod, in as campaignSplitDigest, j as runEval, jt as compareOptimizationMethods, kt as OptimizationTokenUsage, ln as REFERENCE_EQUIVALENCE_INPUT_LIMITS, mn as ReferenceEquivalenceScenario, mt as ExternalTextOptimizerResult, n as SkillOptRunnerCommand, p as LoopProvenanceRecord, pn as ReferenceEquivalenceJudgeResult, pt as ExternalTextOptimizerContext, r as SkillOptTrainerConfig, t as SkillOptOptimizationMethodConfig, un as REFERENCE_EQUIVALENCE_JUDGE_VERSION, ut as composeGate, vn as LlmJudgeOptions, vt as CompareOptimizationMethodsOptions, w as RunImprovementLoopResult, wt as OptimizationMethodProvenance, xt as OptimizationMethodComparison, yn as llmJudge, yt as ComparisonCost, z as OpenAICompatibleOptimizerModel } from "../skillopt-optimization-method-USDKhxSA.js";
|
|
10
10
|
import { a as ObjectiveSource, b as PowerPreflight, c as PromotionPolicy, d as paretoSignificanceGate, i as EvidenceVector, l as buildEvidenceVector, n as AxisVerdict, o as ParetoSignificanceGateOptions, r as BuildEvidenceVectorOptions, s as PromotionObjective, t as AxisEvidence, u as paretoPolicy } from "../promotion-policy-Ckjhzg_4.js";
|
|
11
11
|
import { c as EvalRunGenerationSnapshot, g as TraceSpanEvent, n as HostedTenant, o as EvalRunCellScore, s as EvalRunEvent } from "../client-C9gzZE59.js";
|
|
12
12
|
import { i as InMemoryOutcomeStore, n as FileSystemOutcomeStore, o as OutcomeStore, r as FileSystemOutcomeStoreOptions, t as DeploymentOutcome } from "../outcome-store-BYHIuO0e.js";
|
package/dist/contract/index.js
CHANGED
|
@@ -7,7 +7,7 @@ import { nt as isOtlpModelCall, tt as classifyOtlpSpanRole } from "../kind-facto
|
|
|
7
7
|
import { c as makeProposalFinding } from "../usage-receipt-t7vAzCRQ.js";
|
|
8
8
|
import { a as inMemoryCampaignStorage, i as fsCampaignStorage, r as createRunCostLedger } from "../single-run-lock-DFWHEB09.js";
|
|
9
9
|
import { m as mapConcurrentRange } from "../ledger-core-DXZIqu17.js";
|
|
10
|
-
import { $ as
|
|
10
|
+
import { $ as REFERENCE_EQUIVALENCE_INPUT_LIMITS, F as surfaceHash, P as surfaceContentHash, S as assertOptimizationResult, U as runCampaign, W as resolveRunDir, X as campaignSplitDigest, _ as heldOutGate, a as emitLoopProvenance, b as composeGate, d as runImprovementLoop, dt as llmJudge, et as REFERENCE_EQUIVALENCE_JUDGE_VERSION, g as gepaOptimizationMethod, nt as runReferenceEquivalenceJudge, o as loopProvenanceArgsFromResult, p as runEval, t as skillOptOptimizationMethod, tt as createReferenceEquivalenceJudge, v as defaultProductionGate, w as compareOptimizationMethods, x as externalTextOptimizationMethod } from "../skillopt-optimization-method-CQdVeM8k.js";
|
|
11
11
|
import { E as pairedBootstrap } from "../statistics-ByxzSiOM.js";
|
|
12
12
|
import { i as parseRunRecordSafe, r as modelHasSnapshot } from "../run-record-BmSPWXJR.js";
|
|
13
13
|
import { c as heldoutSignificance, i as powerPreflight, n as paretoPolicy, r as paretoSignificanceGate, t as buildEvidenceVector, u as decidePairedPromotion } from "../promotion-policy-CrLrmys8.js";
|
|
@@ -6,7 +6,7 @@ import { n as CompletionVerdict, o as ProducedState, r as CorrectnessChecker, t
|
|
|
6
6
|
import { D as AnalystIssueExpectation, d as AnalystBenchmarkCase, h as AnalystBenchmarkLabelState, t as AgentProfile$1 } from "./agent-profile-DPi7IZg7.js";
|
|
7
7
|
import { A as LabeledScenarioWrite, D as LabeledScenarioSampleArgs, E as LabeledScenarioRecord, O as LabeledScenarioSource, R as Scenario, S as JudgeConfig, T as LabelTrust, a as CampaignResult, d as DispatchContext, j as MutableSurface, k as LabeledScenarioStore, l as CodeSurface, p as Gate, u as ComponentSurface } from "./types-BnjdJ70P.js";
|
|
8
8
|
import { y as PairedBootstrapResult } from "./statistics-D6Uebe_4.js";
|
|
9
|
-
import { Ft as CampaignRunPlan,
|
|
9
|
+
import { Ft as CampaignRunPlan, Lt as PlanCampaignRunOptions, Zt as CampaignStorage } from "./skillopt-optimization-method-USDKhxSA.js";
|
|
10
10
|
import { N as PairedArmsComparison } from "./statistical-heldout-TQ-4CYiN.js";
|
|
11
11
|
import "./promotion-policy-Ckjhzg_4.js";
|
|
12
12
|
import { D as LedgerHash, m as LedgerTrustedHeadRemoval, p as LedgerTrustedHead } from "./index-CvSN3IG1.js";
|
|
@@ -1512,4 +1512,4 @@ declare function verifyCodeSurface(surface: CodeSurface, worktreeDir?: string):
|
|
|
1512
1512
|
declare function resolveWorktreePath(surface: CodeSurface, worktreeDir?: string): string;
|
|
1513
1513
|
//#endregion
|
|
1514
1514
|
export { SearchPlannedEvent as $, CrossSurfacePairIncompatibilityReason as $n, ToolCallEventLike as $t, SearchAccountingAudit as A, CrossSurfaceAttemptCompleteness as An, UserStory as At, SearchCostAccounting as B, CrossSurfaceCompositionStep as Bn, RunProfileMatrixOptions as Bt, surfaceHash as C, loadEvalFixture as Cn, tangleTracesRoot as Ct, FileSearchLedger as D, AnalyzeCrossSurfaceInteractionsInput as Dn, ScoreboardRenderOptions as Dt, acquireSingleRunLock as E, analyzeCrossSurfaceInteractions as En, PlaybackStep as Et, SearchCandidateRegisteredEvent as F, CrossSurfaceCandidateEvidence as Fn, scoreboardSummary as Ft, SearchLedgerEvent as G, CrossSurfaceInteractionAwareSelection as Gn, BackendIntegrityReport as Gt, SearchLedger as H, CrossSurfaceEligibility as Hn, ScenarioRollup as Ht, SearchCandidateSlot as I, CrossSurfaceCandidateOutcome as In, userStoryScoreboard as It, SearchLedgerTrustedHeadMode as J, CrossSurfaceInteractionReport as Jn, summarizeAgentReceiptIntegrity as Jt, SearchLedgerHash as K, CrossSurfaceInteractionEffect as Kn, assertRealAgentReceipts as Kt, SearchCandidateSlotClosedEvent as L, CrossSurfaceCandidateSummary as Ln, ProfileDispatchFn as Lt, SearchAttemptAccounting as M, CrossSurfaceBootstrapPolicy as Mn, makePlaybackDispatch as Mt, SearchCandidateDecidedEvent as N, CrossSurfaceCandidate as Nn, renderScoreboardMarkdown as Nt, OpenSearchLedgerOptions as O, CrossSurfaceAdditionDecision as On, ScoreboardRow as Ot, SearchCandidateLineage as P, CrossSurfaceCandidateComparison as Pn, scoreUserStory as Pt, SearchPlan as Q, CrossSurfacePairEvidence as Qn, RuntimeEventLike as Qt, SearchCandidateSurface as R, CrossSurfaceComponent as Rn, ProfileMatrixError as Rt, surfaceContentHash as S, discoverEvalFixtures as Sn, resolveRunDir as St, SingleRunLockOptions as T, planEvalFixtureRun as Tn, PlaybackDriver as Tt, SearchLedgerAppendResult as U, CrossSurfaceEvidenceBreakdown as Un, runProfileMatrix as Ut, SearchFailureReason as V, CrossSurfaceDistribution as Vn, RunProfileMatrixResult as Vt, SearchLedgerEntry as W, CrossSurfaceIneligibilityReason as Wn, BackendIntegrityError as Wt, SearchOperationKind as X, CrossSurfaceNaiveStackSelection as Xn, ArtifactEventLike as Xt, SearchModelIdentity as Y, CrossSurfaceInteractionTask as Yn, summarizeBackendIntegrity as Yt, SearchOperationRecordedEvent as Z, CrossSurfacePairCompatibility as Zn, ProposalEventLike as Zt, assertCodeSurfaceIdentity as _, EvalFixtureRunPlan as _n, compareRankKeys as _t, WorktreeAdapterError as a, RolloutArgumentDiff as an, CrossSurfaceTaskRow as ar, SearchSurfaceKind as at, componentSurfaceIdentityMaterial as b, LoadEvalFixtureScenariosOptions as bn, scoreDiscrimination as bt, verifyCodeSurface as c, ScoredRollout as cn, TraceAnalystScenario as cr, SearchTokenAccounting as ct, PhoenixEvaluationResultLike as d, rolloutArgumentDiff as dn, SearchLedgerConflictError as dt, extractProducedState as en, CrossSurfacePairwiseEntry as er, SearchPlannedOperation as et, PhoenixEvaluatorLike as f, NeutralizationGateOptions as fn, SearchLedgerError as ft, isTransientTransportFailure as g, EvalFixtureLoadOptions as gn, campaignMeanComposite as gt, TransientFailureOptions as h, EvalFixtureFile as hn, campaignBreakdown as ht, WorktreeAdapter as i, LabeledScenarioStoreError as in, CrossSurfaceSelections as ir, SearchSurfaceEvidence as it, SearchArtifactRef as j, CrossSurfaceBestSingleSelection as jn, UserStoryVerdict as jt, SEARCH_LEDGER_SCHEMA as k, CrossSurfaceAdditionRejectionReason as kn, ScoreboardSummary as kt, AutoevalsScoreLike as l, UngroundedLiteralReport as ln, buildTraceAnalystSurfaceDispatch as lr, openSearchLedger as lt, phoenixEvaluatorJudge as m, EvalFixture as mn, CampaignBreakdown as mt, GitWorktreeAdapterOptions as n, FsLabeledScenarioStore as nn, CrossSurfaceRelativeCost as nr, SearchSourceRef as nt, gitWorktreeAdapter as o, RolloutArgumentDiffOptions as on, BuildTraceAnalystSurfaceDispatchOptions as or, SearchTaskAttemptedEvent as ot, autoevalsScorerJudge as p, neutralizationGate as pn, SearchLedgerIntegrityError as pt, SearchLedgerReplay as q, CrossSurfaceInteractionPath as qn, assertRealBackend as qt, Worktree as r, FsLabeledScenarioStoreOptions as rn, CrossSurfaceSelectionPolicy as rr, SearchSurfaceEffect as rt, resolveWorktreePath as s, RolloutCall as sn, TraceAnalystArtifact as sr, SearchTaskOutcome as st, CodeSurfaceVerification as t, neutralizeText as tn, CrossSurfaceRankedSingle as tr, SearchPlannedTask as tt, AutoevalsScorerLike as u, classifyUngroundedLiterals as un, traceAnalystQualityJudge as ur, validateSearchLedgerEvent as ut, assertComponentSurface as v, EvalFixtureScenario as vn, DiscriminationScore as vt, SingleRunLock as w, loadEvalFixtureScenarios as wn, PlaybackContext as wt, renderSurfaceDiff as x, PlanEvalFixtureRunOptions as xn, selectDiscriminative as xt, codeSurfaceIdentityMaterial as y, EvalFixtureValidationMode as yn, ScenarioSignal as yt, SearchCompletedEvent as z, CrossSurfaceComponentEvidence as zn, ProfileSummary as zt };
|
|
1515
|
-
//# sourceMappingURL=index-
|
|
1515
|
+
//# sourceMappingURL=index-DPPGNJ_R.d.ts.map
|
|
@@ -1 +1 @@
|
|
|
1
|
-
{"version":3,"file":"index-
|
|
1
|
+
{"version":3,"file":"index-DPPGNJ_R.d.ts","names":[],"sources":["../src/campaign/analyst-surface.ts","../src/campaign/cross-surface-types.ts","../src/campaign/cross-surface-interaction.ts","../src/campaign/fixtures.ts","../src/campaign/gates/neutralization-gate.ts","../src/campaign/grounded-reflection.ts","../src/campaign/labeled-store/fs-adapter.ts","../src/campaign/neutralize.ts","../src/produced-state.ts","../src/integrity/backend-integrity.ts","../src/campaign/presets/run-profile-matrix.ts","../src/campaign/presets/playback.ts","../src/campaign/run-dir.ts","../src/campaign/scenario-selection.ts","../src/campaign/score-utils.ts","../src/campaign/search-ledger-errors.ts","../src/campaign/search-ledger.ts","../src/campaign/single-run-lock.ts","../src/campaign/surface-identity.ts","../src/campaign/transient-failure.ts","../src/campaign/upstream-evaluators.ts","../src/campaign/worktree/index.ts"],"mappings":";;;;;;;;;;;;;;UAWiB,6BAA6B;EAC5C;EACA,YAAY;EACZ,YAAY;EACZ,yBAAyB;EACzB,kBAAkB;;UAGH;EACf,mBAAmB;EACnB,QAAQ;EACR,MAAM;;UAGS;EACf,QAAQ;IACN;IACA,YAAY;IACZ;IACA,QAAQ;MACN,QAAQ;;iBAGE,iCACd,SAAS,2CAET,SAAS,gBACT,UAAU,sBACV,SAAS,oBACN,QAAQ;iBAgBG,4BAA4B,YAC1C,sBACA;;;;KCtDU;;UAGK;EACf;EACA;;EAEA;;;UAIe;EACf;EACA;EACA;EACA;;;UAIe;EACf;;EAEA;;EAEA;;;;;;;UAQe;EACf;EACA;;EAEA;EACA,cAAc;EACd;EACA;;;;;;EAMA,MAAM;EACN,mBAAmB;;EAEnB;;UAGe;EACf;EACA;EACA;;;UAIe;EACf;EACA;EACA;EACA;;EAEA,kCAAkC;;EAElC;;UAGe,qCACf,aAAa,sBAAsB;EAEnC,qBAAqB;EACrB,qBAAqB;EACrB,eAAe;EACf;;EAEA;;EAEA;;EAEA;;EAEA;EACA,WAAW;EACX,WAAW;;UAGI;EACf;EACA;EACA;EACA;EACA;EACA;;UAGe;EACf;EACA;EACA;EACA;;UAGe;EACf,aAAa;EACb;EACA;EACA;EACA;;KAGU;UAWK;EACf;EACA,SAAS;;UAGM;EACf;EACA;EACA;EACA;EACA;EACA;EACA;EACA;;UAGe;EACf,WAAW;EACX,SAAS;EACT,OAAO;EACP,OAAO,eAAe;EACtB,QAAQ;EACR,QAAQ;;EAER,sBAAsB;;EAEtB,aAAa;;UAGE;EACf;EACA;EACA;;EAEA;;UAGe;EACf;EACA;EACA;EACA;EACA;EACA,QAAQ;EACR,cAAc,eAAe;;UAGd;EACf;EACA;EACA;EACA;EACA;;UAGe;EACf;;EAEA;EACA;;UAGe;EACf,SAAS;EACT;EACA;EACA;EACA;EACA,eAAe;EACf,gBAAgB;;KAGN;UAYK;EACf;EACA,SAAS;EACT;;UAGe;EACf;EACA;EACA;EACA;EACA;EACA;EACA;EACA,4BAA4B,iCAAiC;EAC7D,wBAAwB,eAAe;EACvC,QAAQ;EACR,QAAQ;EACR,aAAa;EACb,eAAe;;UAGA;EACf;EACA;EACA;;UAGe;EACf;EACA;EACA,SAAS;;UAGM;;EAEf;EACA;;KAGU;UAaK;EACf;EACA;EACA;EACA;EACA;EACA,uBAAuB;EACvB;EACA;EACA,SAAS;;UAGM;EACf;EACA;EACA,YAAY;EACZ;;;UAIe;EACf;EACA;EACA;EACA;EACA,OAAO;;UAGQ;;EAEf;;EAEA;EACA;;EAEA;EACA;;EAEA,gBAAgB;;EAEhB,OAAO;;UAGQ;EACf,YAAY;EACZ,YAAY;EACZ,kBAAkB;;UAGH,8BACf,aAAa,sBAAsB;EAEnC;EACA;EACA;EACA;;EAEA,MAAM;EACN,iBAAiB;EACjB,iBAAiB;EACjB,YAAY;EACZ,UAAU;EACV,YAAY;;;;;;;;;iBCnRE,gCAAgC,aAAa,qBAC3D,OAAO,qCAAqC,QAC3C,8BAA8B;;;KC7CrB;UAEK;EACf;EACA;EACA;;UAGe;EACf;EACA;EACA;EACA;EACA;EACA;EACA,OAAO;EACP;;UAGe,4BAA4B;EAC3C;EACA;EACA;EACA;EACA;EACA;EACA;EACA;;UAGe;;EAEf,aAAa;;EAEb;;UAGe,wCAAwC;EACvD;;UAGe,0BAA0B,6BACjC,KACN,uBAAuB,qBAAqB;EAG9C;EACA,aAAa;EACb;EACA;EACA,UAAU;;KAGA,qBAAqB;EAC/B,UAAU,MAAM,KAAK;;;iBAIP,qBAAqB;;;;;iBA4BrB,gBACd,kBACA,cACA,UAAS,yBACR;;iBA4Ca,yBACd,kBACA,UAAS,kCACR;;;;;iBAuBa,mBAAmB,qBACjC,SAAS,0BAA0B,aAClC;;;UChJc,0BAA0B,kBAAkB,WAAW;EACtE,WAAW;;;;;EAKX;;;;;;iBAuBc,mBAAmB,WAAW,kBAAkB,UAC9D,SAAS,0BAA0B,aAClC,KAAK,WAAW;;;;;;;;;;;;;;;;;;;;;;;;;;;;;UC9BF;WACN;WACA,MAAM,SAAS;;;UAIT;;WAEN;;WAEA;WACA,gBAAgB;;UAGV;;WAEN;;WAEA;;UAGM;;WAEN;;WAEA,eAAe;;WAEf,eAAe;;;;;;;;iBASV,oBACd,mBAAmB,iBACnB,OAAM,6BACL;UAsCc;;WAEN;;WAEA;;;;;;;;;iBAUK,2BACd,cACA,MAAM,KAAK,0DACV;;;UCjFc;;EAEf;;;EAGA;;EAEA;;;cAIW,kCAAkC;WAE3B;EADlB,YACkB,cAChB;;;;;;cAiBS,kCAAkC;mBAIhB;mBAHZ;mBACA;EAEjB,YAA6B,SAAS;EAKhC,QAAQ,OAAO,uBAAuB;EAYtC,OAAO,MAAM,4BAA4B,QAAQ;EA8CjD,QAAQ;IACZ;IACA;IACA,UAAU;IACV,SAAS,OAAO;;UAkCV;UAiCA;UAmBA;UAkBA;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;iBCtNM,eAAe;;;;UCZd;EACf;EACA;;;;;;;UAQe;EACf;EACA;EACA;EACA;EACA;EACA;;;UAIe;EACf;EACA;EACA;EACA;EAIA;;;;;;;;KASU,mBACR,oBACA,oBACA;EACE;;;;;;;;;;;;iBAmBU,qBAAqB,iBAAiB,qBAAqB;;;UCpD1D;;EAEf;;EAEA;;EAEA;;EAEA;;EAEA;;EAEA;;EAEA;;EAEA;;EAEA;;;;;;;cAQW,8BAA8B;WAGvB,QAAQ;EAF1B,YACE,iBACgB,QAAQ;;;;;;;iBAWZ,0BACd,SAAS,cAAc,aACtB;;iBAWa,+BACd,UAAU,cAAc,eACvB;;;;;;;;iBA0Ga,kBACd,SAAS,cAAc,YACvB;EAAQ;IACP;;iBAMa,wBACd,UAAU,cAAc,cACxB;EAAQ;IACP;;;;;;cC7HU,2BAA2B;EACtC,YAAY;;;;;KAQF,kBAAkB,kBAAkB,UAAU,cACxD,SAAS,gBACT,UAAU,WACV,KAAK,oBACF,QAAQ;UAEI,wBAAwB,kBAAkB,UAAU;;EAEnE,UAAU;;EAEV,WAAW;;EAEX,UAAU,kBAAkB,WAAW;;EAEvC,SAAS,YAAY,WAAW;;;EAGhC;;;EAGA;;;EAGA;;;;EAIA;;EAEA,WAAW;;EAEX;;EAEA;;;;;;;;EAQA;;;EAGA;;EAEA;;;EAGA;;EAEA;;EAEA,eAAe;EACf,gBAAgB;;;EAGhB,UAAU;;EAEV,YAAY;;;EAGZ,aAAa,UAAU;;;;EAIvB;;;;;;;EAOA,cACE,UAAU,WACV,UAAU;IACL;IAAgB;;;UAGR;EACf;EACA;EACA;;EAEA;;EAEA;;EAEA;EACA,gBAAgB;;;EAGhB,WAAW;;UAGI;EACf;EACA;;UAGe,uBAAuB,WAAW,kBAAkB;EACnE;EACA;;;;EAIA,SAAS;EACT,WAAW,eAAe;EAC1B,YAAY,eAAe;;EAE3B,YAAY,eAAe;;EAE3B,WAAW;;EAEX,WAAW,eAAe,eAAe,WAAW;;;;;iBA2IhC,iBAAiB,kBAAkB,UAAU,WACjE,MAAM,wBAAwB,WAAW,aACxC,QAAQ,uBAAuB,WAAW;;;;;UCzT5B;;EAEf;;EAEA,UAAU;;;;;;;UAQK,kBAAkB;;EAEjC;;EAEA,OAAO;;EAEP,cAAc;;;UAIC,wBAAwB;EACvC,SAAS;;;;;;;;;;;UAYM,eAAe,eAAe,YAAY;EACzD,IAAI,OAAO,QAAQ,KAAK,kBAAkB,iBAAiB;;;;;;;iBAQ7C,qBAAqB,eAAe,WAClD,QAAQ,eAAe,UACtB,kBAAkB,QAAQ;;UAQZ,yBAAyB;EACxC;;;;;;;;iBASoB,eACpB,OAAO,WACP,OAAO,eACP,kBAAkB,qBACjB,QAAQ;;UAUM;EACf;EACA;EACA;EACA;EACA;EACA;;;;;;;iBAQc,oBAAoB,mBAAmB,qBAAqB;;UAkB3D;;EAEf;;EAEA;;EAEA;;EAEA;;EAEA;;EAEA;;;iBAIc,kBAAkB,eAAe,kBAAkB;UAwBlD;;EAEf;;EAEA,OAAO;;EAEP;;;;;;;;;iBAkBc,yBACd,eAAe,iBACf,OAAM;;;;;;;;iBCnMQ;;;;;iBAQA,cAAc,gBAAgB;;;;;;;;;;;;;;;;;UCD7B;EACf;;EAEA;;UAGe;EACf;;EAEA;;EAEA;EACA;;EAEA;;;;;;;;;;iBA4Cc,oBACd,SAAS,kBACT;EAAS;IACR;;;;;;;;;;;iBA0Ba,qBACd,SAAS,kBACT,WACA;EAAS;;;;;;;iBC3FK,sBAAsB,WAAW,kBAAkB,UACjE,UAAU,eAAe,WAAW;;;;iBA0CtB,gBAAgB,sBAAsB;UAWrC;;EAEf,YAAY;;;;;EAKZ,WAAW;IAAQ;IAAoB;IAAmB;IAAgB;;;;;iBAK5D,kBAAkB,WAAW,kBAAkB,UAC7D,UAAU,eAAe,WAAW,aACnC;;;;cC/EU,0BAA0B;;cAG1B,mCAAmC;;cAGnC,kCAAkC;;;cCgClC;KAED,mBAAmB;KAEnB;;;UAYK;EACf;EACA;EACA,QAAQ;EACR;;;;UAKe;EACf;EACA;;UAGe;EACf;EACA;;UAGe;EACf;EACA,MAAM;EACN,UAAU;;UAGK;;EAEf;EACA;EACA;EACA;EACA,gBAAgB;;KAGN;UAOK;EACf;EACA,QAAQ;EACR,WAAW;;;EAGX;;UAGe;EACf;EACA,MAAM;;UAGS;EACf;;;EAGA;;UAGe;;EAEf,gBAAgB;;EAEhB,OAAO;;EAEP,YAAY;;KAGF;EAEN;EACA;EACA;EACA;;EAGA;EACA;;KAGM;EAEN;EACA;EACA;;EAGA;;EAEA;EACA;;UAGW;EACf,QAAQ;EACR,MAAM;;UAGS;EACf;EACA;;KAGU;EAEN;EACA;EACA,SAAS;;EAGT;EACA;EACA,SAAS;EACT,SAAS;;EAGT;EACA,SAAS;EACT,OAAO;IAAwB;;;KAGzB;EAEN;EACA;EACA;EACA;EACA;;EAGA;EACA;;;;UAKW;EACf;EACA;EACA;EACA,QAAQ;EACR,UAAU;;UAGF;EACR;EACA;EACA,WAAW;;UAGI,2BAA2B;EAC1C;EACA,MAAM;;UAGS,uCAAuC;EACtD;EACA;EACA;EACA;EACA,SAAS;EACT,UAAU;;UAGK,uCAAuC;EACtD;EACA;EACA;EACA,QAAQ;;UAGO,iCAAiC;EAChD;EACA;EACA;EACA;EACA;IACE;IACA,QAAQ;;EAEV;IACE,OAAO;IACP,OAAO;IACP,WAAW;;EAEb,SAAS;EACT,YAAY;EACZ,iBAAiB;;UAGF,qCAAqC;EACpD;EACA;EACA,eAAe;EACf;IAEM;IACA,OAAO;IACP,QAAQ;;IAGR;IACA,QAAQ;;EAEd;IACM;;IACA;IAAmB,SAAS;;IAC5B;IAAkB,SAAS;;EACjC,YAAY;;UAGG,oCAAoC;EACnD;EACA;EACA;IACM;;IAEA;IACA,QAAQ;;;UAIC,6BAA6B;EAC5C;EACA;IAEM;IACA;;IAGA;IACA,QAAQ;;;KAIJ,oBACR,qBACA,iCACA,iCACA,2BACA,+BACA,8BACA;UAEa;EACf,eAAe;EACf;EACA;EACA,cAAc;EACd,OAAO;EACP,WAAW;;KAGD;EAEN;EACA;EACA;EACA;EACA;;EAGA;EACA;EACA;EACA;EACA;EACA;EACA;;UAGW;EACf;EACA;EACA;EACA;EACA;EACA;EACA;IAAY;IAAgB;IAAgB;;EAC5C;IAAqB;IAAmB;IAAiB;;EACzD;IAAa;IAAkB;IAAkB;;EACjD;IACE;IACA;IACA;IACA;IACA;IACA;;EAEF;EACA;EACA,YAAY;EACZ,UAAU;;UAGK;EACf,SAAS;EACT,MAAM;EACN,YAAY;EACZ,sBAAsB;EACtB,UAAU;EACV,YAAY;EACZ,WAAW;EACX,YAAY;EACZ,OAAO;;UAGQ;EACf,OAAO;;EAEP;EACA,QAAQ;;;;iBA+ZM,0BAA0B,iBAAiB;;;;;;;;;;;;;;;KAsB/C;UAEK;EACf;EACA;EACA,cAAc;;UAGC;WACN;WACA;;WAEA;EACT,OAAO,OAAO,oBAAoB,QAAQ;EAC1C,UAAU,QAAQ;;EAElB,eAAe,QAAQ;;;EAGvB,kBAAkB,QAAQ;;;;;;;EAO1B,oBAAoB,QAAQ;;;;iBAKd,iBAAiB,SAAS,0BAA0B;;cA8BvD,4BAA4B;WAC9B;WACA;WACA;mBACQ;mBACA;EAMjB,YAAY,cAAc,oBAAoB,cAAa;EAerD,UAAU,QAAQ;EAIlB,OAAO,OAAO,oBAAoB,QAAQ;EAU1C,eAAe,QAAQ;EAIvB,kBAAkB,QAAQ;EAI1B,oBAAoB,QAAQ;;;;;;;;;;;;;;;;;;;UC/3BnB;;WAEN;;WAEA;;WAEA;;WAEA;;UAGM;;EAEf;;;;;;;iBAwBc,qBAAqB,MAAM,uBAAuB;;;;iBCpDlD,0BAA0B,2BAA2B,WAAW;;iBAmChE,uBAAuB,2BAA2B,WAAW;;iBA8B7D,iCAAiC,SAAS;;;;iBAa1C,4BAA4B,SAAS;;iBAgBrC,mBAAmB,SAAS;;iBAW5B,YAAY,SAAS;;iBAKrB,kBACd,eAAe,gBACf,iBAAiB;;;;;;;;;;;;;;;;;;;;;UCpGF;;;;;;;WAON;;WAEA,yBAAyB;;;;;;iBAYpB,4BACd,oCACA,OAAM;;;UCnCS;EACf;EACA;EACA;;UAGe,qBAAqB,gBAAgB;EACpD;EACA;EACA;EACA,SACE,QAAQ,SACR,SAAS,4BACR,QAAQ;;UAGI;EACf;EACA;EACA,WAAW;;KAGD,oBAAoB,eAAe,4BAC7C,OAAO,QACP,SAAS,8BACN,qBAAqB,QAAQ;UAEjB;WACN,QAAQ;WACR;;KAGN,sBAAsB,WAAW,KACpC,iBAAiB;EAGjB;;UAGQ,qBAAqB,kBAAkB;EAC/C;EACA;EACA;EACA,aAAa,UAAU;;EAEvB,eAAe;;iBAGD,sBACd,gBAAgB,yBAChB,WACA,kBAAkB,WAAW,UAE7B,WAAW,qBAAqB,UAChC,SAAS,qBAAqB;EAC5B,SAAS;IAAS,UAAU;IAAW,UAAU;MAAc;EAC/D,WAAW,sBAAsB;IAElC,YAAY,WAAW;iBA8CV,qBACd,eAAe,yBACf,WACA,kBAAkB,WAAW,UAE7B,QAAQ,oBAAoB,SAC5B,SAAS,qBAAqB;EAC5B;EACA,SAAS;IAAS,UAAU;IAAW,UAAU;MAAc;;EAEzD;EAAc;;EACd;EAAa,UAAU,sBAAsB;KAEpD,YAAY,WAAW;;;KCxFrB,qBAAqB;KACrB,iBAAiB,SAAS;KAC1B,aAAa,gBAAgB,aAAa,MAAM,mBAAmB;UAcvD;;WAEN;;WAEA;;WAEA;;WAEA;;WAEA;;UAGM;;EAEf,OAAO;IAAQ;IAAiB;MAAkB,QAAQ;;;EAG1D,SAAS,UAAU,UAAU,kBAAkB,QAAQ;;EAEvD,QAAQ,UAAU,WAAW;;;cAIlB,6BAA6B;WAG7B;EAFX,YACE,iBACS;;UAOI;;EAEf;;EAEA;;EAEA;;;;EAIA,MAAM;;UAobS;;EAEf;;EAEA;;EAEA;;;EAGA,YAAY;;;;;iBAoJE,mBAAmB,MAAM,4BAA4B;;;;iBA0FrD,kBACd,SAAS,aACT,uBACC;;;iBAOa,oBAAoB,SAAS,aAAa"}
|
|
@@ -1,7 +1,7 @@
|
|
|
1
1
|
import { s as RunSplitTag } from "./run-record-CF4Dwpxr.js";
|
|
2
2
|
import { R as Scenario, a as CampaignResult, d as DispatchContext } from "./types-BnjdJ70P.js";
|
|
3
|
-
import {
|
|
4
|
-
import "./index-
|
|
3
|
+
import { Zt as CampaignStorage } from "./skillopt-optimization-method-USDKhxSA.js";
|
|
4
|
+
import "./index-DPPGNJ_R.js";
|
|
5
5
|
//#region src/benchmarks/types.d.ts
|
|
6
6
|
type BenchmarkTaskKind = 'retrieval' | 'rag-answer' | 'hallucination' | 'kb-improvement' | 'routing' | 'custom';
|
|
7
7
|
type BenchmarkFamily = 'beir' | 'mteb-retrieval' | 'msmarco' | 'trec-dl' | 'miracl' | 'lotte' | 'bright' | 'crag' | 'hotpotqa' | 'kilt' | 'ragtruth' | 'faithbench' | 'first-party' | 'custom';
|
|
@@ -332,4 +332,4 @@ declare namespace index_d_exports {
|
|
|
332
332
|
}
|
|
333
333
|
//#endregion
|
|
334
334
|
export { BenchmarkMetricCalibrationOptions as A, BenchmarkSource as B, BenchmarkRunOptions as C, runBenchmarkAdapter as D, renderBenchmarkReportMarkdown as E, BenchmarkDatasetItem as F, deterministicSplit as H, BenchmarkEvaluation as I, BenchmarkFamily as L, calibrateBenchmarkMetric as M, BENCHMARK_SPLIT_SEED as N, summarizeBenchmarkCampaign as O, BenchmarkAdapter as P, BenchmarkResponder as R, BenchmarkReport as S, BenchmarkSliceSummary as T, BenchmarkTaskKind as V, parseJsonlRows as _, StandardRetrievalDocument as a, retrievalMetricsAtCutoff as b, StandardRetrievalQrel as c, buildStandardRetrievalItems as d, createRetrievalIdBenchmarkAdapter as f, parseBeirQueriesJsonl as g, parseBeirCorpusJsonl as h, StandardRetrievalArtifact as i, BenchmarkMetricCalibrationResult as j, index_d_exports$1 as k, StandardRetrievalQuery as l, normalizeRetrievedDocumentIds as m, BuildStandardRetrievalItemsOptions as n, StandardRetrievalEvaluationOptions as o, evaluateStandardRetrieval as p, RetrievalIdAdapterOptions as r, StandardRetrievalPayload as s, index_d_exports as t, StandardRetrievalResult as u, parseQrels as v, BenchmarkRunResult as w, BenchmarkDistribution as x, parseTsvRows as y, BenchmarkScenario as z };
|
|
335
|
-
//# sourceMappingURL=index-
|
|
335
|
+
//# sourceMappingURL=index-YE4KdKbO2.d.ts.map
|
|
@@ -1 +1 @@
|
|
|
1
|
-
{"version":3,"file":"index-
|
|
1
|
+
{"version":3,"file":"index-YE4KdKbO2.d.ts","names":[],"sources":["../src/benchmarks/types.ts","../src/benchmarks/calibration.ts","../src/benchmarks/routing/dataset.ts","../src/benchmarks/routing/index.ts","../src/benchmarks/runner.ts","../src/benchmarks/standard-formats.ts","../src/benchmarks/index.ts"],"mappings":";;;;;KASY;KAQA;UAgBK,qBAAqB;;;EAGpC;;EAEA,SAAS;;EAET,QAAQ;;EAER,SAAS;;EAET,WAAW;;EAEX;;EAEA,SAAS;EACT,WAAW;;UAGI;;;;EAIf;;EAEA;;EAEA,aAAa;;;EAGb,KAAK;EACL;;UAGe;EACf;EACA;EACA;EACA;EACA;;;UAQe,iBAAiB,kBAAkB,oBAAoB;;EAEtE;EACA,SAAS;EACT,WAAW;EACX;EACA,SAAS;EACT;;;;;EAKA,YAAY,OAAO,cAAc,QAAQ,qBAAqB;;EAE9D,SAAS,MAAM,qBAAqB,WAAW,UAAU,YAAY,QAAQ;;;;EAI7E,YAAY,iBAAiB;;UAGd,kBAAkB,4BAA4B;EAC7D;EACA;EACA,QAAQ;EACR,UAAU;EACV,UAAU;EACV,MAAM,qBAAqB;;KAGjB,mBAAmB,oBAAoB,uBAAuB;EACxE,UAAU,kBAAkB;EAC5B,MAAM,qBAAqB;EAC3B,SAAS;MACL,QAAQ,aAAa;;;cAoBd;;;;;;;;;iBAUG,mBACd,gBACA,gBACC;;;UCjJc,kCAAkC,oBAAoB;EACrE,SAAS,iBAAiB,qBAAqB,WAAW,UAAU;EACpE,MAAM,qBAAqB;EAC3B,cAAc;EACd,gBAAgB;EAChB;EACA;EACA;;UAGe;EACf;EACA,MAAM;EACN,QAAQ;EACR;EACA;EACA;EACA;;iBAGoB,yBAAyB,oBAAoB,oBACjE,SAAS,kCAAkC,UAAU,aACpD,QAAQ;;;;;;;;;;;;;;;;;;;;;;;;UCFM;EACf;EACA;EACA;;EAEA;;EAEA;;EAEA;;cAGW,iBAAiB;;;;KChBlB,iBAAiB;KACjB,qBAAqB,qBAAqB;cAEhD,0BAA0B,iBAAiB,oBAAoB;WAC1D;WACA;WACA;WACA;WACA;EAEH,YAAY,OAAO,cAAc,QAAQ;EAMzC,SAAS,MAAM,oBAAoB,mBAAmB,QAAQ;EAqBpE,YAAY,iBAAiB;;;;;;;;iBAef,mBAAmB;cAOtB,cAAW,OAjDG,gBAAc,QAAQ;cAkDpC,WAAQ,MA5CE,oBAAkB,qBAAqB,QAAQ;cA6CzD,cAAW,mBAxBO;;;UClCd,oBAAoB,oBAAoB;EACvD,SAAS,iBAAiB,qBAAqB,WAAW,UAAU;EACpE,SAAS,mBAAmB,UAAU;EACtC,kBAAkB;EAClB;EACA;EACA;EACA;EACA;EACA;EACA;EACA;EACA;EACA,UAAU;EACV,YAAY;;UAGG;EACf;EACA,QAAQ;EACR,UAAU;EACV,SAAS;EACT;EACA;EACA;EACA;EACA;EACA;EACA;EACA;EACA;EACA;EACA;EACA,QAAQ,eAAe;EACvB,MAAM,eAAe;EACrB,YAAY,eAAe;EAC3B,OAAO;EACP,SAAS;EACT,WAAW;;UAGI;EACf;EACA;EACA;EACA,OAAO;EACP,SAAS;EACT,WAAW;;UAGI;EACf;EACA;EACA;EACA;EACA;EACA;;UAGe,mBAAmB,oBAAoB;EACtD,WAAW,MAAM,kBAAkB;EACnC,UAAU,eAAe,WAAW,kBAAkB;EACtD,QAAQ;EACR;EACA;;iBAGoB,oBAAoB,oBAAoB,oBAC5D,SAAS,oBAAoB,UAAU,aACtC,QAAQ,mBAAmB,UAAU;iBAmGxB,2BAA2B,UAAU,WAAW;EAC9D,SAAS,iBAAiB,qBAAqB,WAAW,UAAU;EACpE,WAAW,MAAM,kBAAkB;EACnC,UAAU,eAAe,WAAW,kBAAkB;IACpD;iBA6CY,8BAA8B,QAAQ;;;UCpOrC;EACf;EACA;EACA;EACA,WAAW;;UAGI;EACf;EACA;EACA,WAAW;;UAGI;EACf;EACA;EACA;;UAGe;EACf;EACA;EACA;EACA,gBAAgB;EAChB,SAAS,eAAe;EACxB,WAAW;;UAGI;EACf;EACA,QAAQ;EACR,kBAAkB;EAClB,gBAAgB;EAChB,kBAAkB;EAClB;EACA,SAAS;EACT;EACA,WAAW,oBAAoB;;UAGhB,kCAAkC;EACjD,oBAAoB;EACpB;EACA;EACA;EACA;;UAGe;EACf;EACA;EACA;EACA;;KAGU,kEAGC;EAEP;EACA;EACA,mBAAmB;;UAGR;EACf,oBAAoB;EACpB;EACA;EACA;EACA;;iBAGc,eAAe,aAAa,eAAe;iBAc3C,aAAa;iBASb,WAAW,eAAe;iBAgB1B,qBAAqB,eAAe;iBAgBpC,sBAAsB,eAAe;iBAerC,4BACd,SAAS,qCACR,MAAM,qBAAqB;iBAuCd,kCACd,SAAS,4BACR,iBACD,qBAAqB,2BACrB,0BACA;iBAqBc,0BACd,SAAS,0BACT,UAAU,2BACV,UAAS;;;;;;;;;;iBAsCK,8BACd,UAAU,2BACV,oBAAmB;iBA0BL,yBAAyB;EACvC;EACA,gBAAgB;EAChB;IACE"}
|
package/dist/index.d.ts
CHANGED
|
@@ -28,9 +28,9 @@ import { a as FailureClassification, c as classifyFailure, i as DEFAULT_RULES, o
|
|
|
28
28
|
import { C as toOpenAiTool, S as makeEvalTools, _ as TraceAnalysisToolDescriptor, a as TraceAnalystLimits, b as EvalToolDef, d as RawAnalystFinding, g as TRACE_ANALYST_TOOL_NAMESPACE, h as BuildTraceAnalysisToolsOptions, i as TraceAnalysisEngineResult, l as RawAnalystEvidence, n as TraceAnalysisEngine, o as resolveTraceAnalystLimits, r as TraceAnalysisEngineRequest, t as DEFAULT_TRACE_ANALYST_LIMITS, v as buildTraceAnalysisToolDescriptors, x as MakeEvalToolsConfig, y as traceAnalystFunctionGroup } from "./engine-3hL-XqwJ.js";
|
|
29
29
|
import { A as isTrainableSplit, D as assertRolloutLine, E as assertMintedLines, T as assertMinted, b as RolloutSplit, h as RolloutLine, i as ChatToolCall, j as validateRolloutLine, k as isRolloutLine, n as ChatMessage, o as MintedRolloutLine, p as RolloutCapture, s as MintedRolloutOutcome, u as ROLLOUT_SCHEMA, w as ToolDef, x as RolloutStep, y as RolloutRole } from "./schema-Cef2cFmb2.js";
|
|
30
30
|
import { C as Unavailable, D as showMeasured, E as isUnavailable, S as SupervisorRunTreeGapCode, _ as SupervisorRunReport, b as SupervisorRunTree, f as SUPERVISOR_RUN_SCHEMA, g as SupervisorRunReader, h as SupervisorRunNodeRole, p as SourceLimits, r as Measured, v as SupervisorRunRollup, x as SupervisorRunTreeGap, y as SupervisorRunSources } from "./types-D4mog56g.js";
|
|
31
|
-
import { B as BenchmarkSource, F as BenchmarkDatasetItem, H as deterministicSplit, I as BenchmarkEvaluation, L as BenchmarkFamily, N as BENCHMARK_SPLIT_SEED, P as BenchmarkAdapter, R as BenchmarkResponder, V as BenchmarkTaskKind, t as index_d_exports, z as BenchmarkScenario } from "./index-
|
|
32
|
-
import { $ as redTeamDataset, J as RedTeamCase, Q as RedTeamReport, X as RedTeamFinding, Y as RedTeamCategory, Z as RedTeamPayload,
|
|
33
|
-
import { $t as ToolCallEventLike, Gt as BackendIntegrityReport, Jt as summarizeAgentReceiptIntegrity, Kt as assertRealAgentReceipts, Qt as RuntimeEventLike, Wt as BackendIntegrityError, Xt as ArtifactEventLike, Yt as summarizeBackendIntegrity, Zt as ProposalEventLike, en as extractProducedState, qt as assertRealBackend } from "./index-
|
|
31
|
+
import { B as BenchmarkSource, F as BenchmarkDatasetItem, H as deterministicSplit, I as BenchmarkEvaluation, L as BenchmarkFamily, N as BENCHMARK_SPLIT_SEED, P as BenchmarkAdapter, R as BenchmarkResponder, V as BenchmarkTaskKind, t as index_d_exports, z as BenchmarkScenario } from "./index-YE4KdKbO2.js";
|
|
32
|
+
import { $ as redTeamDataset, J as RedTeamCase, Q as RedTeamReport, X as RedTeamFinding, Y as RedTeamCategory, Z as RedTeamPayload, _n as LlmJudgeDimension, at as CanaryKind, ct as CanarySeverity, dn as ReferenceEquivalenceJudgeInput, et as redTeamReport, fn as ReferenceEquivalenceJudgeOptions, gn as runReferenceEquivalenceJudge, hn as createReferenceEquivalenceJudge, it as CanaryEvaluation, ln as REFERENCE_EQUIVALENCE_INPUT_LIMITS, lt as runCanaries, mn as ReferenceEquivalenceScenario, nt as toolNamesForRun, ot as CanaryOptions, pn as ReferenceEquivalenceJudgeResult, q as DEFAULT_RED_TEAM_CORPUS, rt as CanaryAlert, st as CanaryReport, tt as scoreRedTeamOutput, un as REFERENCE_EQUIVALENCE_JUDGE_VERSION, vn as LlmJudgeOptions, yn as llmJudge } from "./skillopt-optimization-method-USDKhxSA.js";
|
|
33
|
+
import { $t as ToolCallEventLike, Gt as BackendIntegrityReport, Jt as summarizeAgentReceiptIntegrity, Kt as assertRealAgentReceipts, Qt as RuntimeEventLike, Wt as BackendIntegrityError, Xt as ArtifactEventLike, Yt as summarizeBackendIntegrity, Zt as ProposalEventLike, en as extractProducedState, qt as assertRealBackend } from "./index-DPPGNJ_R.js";
|
|
34
34
|
import { A as PairArmsResult, C as hashJson, D as MatchedPair, E as ComparePairedArmsOptions, F as PairedMetricDelta, I as comparePairedArms, L as pairArms, M as PairedArmRow, N as PairedArmsComparison, O as MatchedRunRecordPair, P as PairedCorrectness, R as pairRunRecords, S as evaluateHypothesis, T as verifyManifest, _ as HypothesisManifest, b as SignedManifestAlgo, j as PairRunRecordsResult, k as PairArmsOptions, v as HypothesisResult, w as signManifest, x as canonicalize, y as SignedManifest } from "./statistical-heldout-TQ-4CYiN.js";
|
|
35
35
|
import { _ as paretoFrontier, f as Direction, g as dominates, h as crowdingDistance, m as ParetoResult, p as Objective, v as paretoFrontierWithCrowding, y as scalarScore } from "./promotion-policy-Ckjhzg_4.js";
|
|
36
36
|
import { _ as vitestTestParser, a as DockerSandboxDriver, c as SandboxHarness, d as SubprocessSandboxDriver, f as SubprocessSandboxDriverOptions, g as pytestTestParser, h as jestTestParser, i as runTestGradedScenario, l as SandboxHarnessResult, m as composeParsers, n as TestGradedRunResult, o as HarnessConfig, p as TestOutputParser, r as TestGradedScenario, s as SandboxDriver, t as TestGradedRunOptions, u as SandboxResult } from "./test-graded-scenario-D1TaI2va.js";
|
package/dist/index.js
CHANGED
|
@@ -19,7 +19,7 @@ import { a as RunCritic, d as defaultIsMaterial, f as diffFindings, h as defineT
|
|
|
19
19
|
import { a as inMemoryVerdictCache, i as fileVerdictCache, n as canonicalJson, r as contentHash, t as cachedJudge } from "./verdict-cache-BCcOh0kF.js";
|
|
20
20
|
import { t as createDspyRlmTraceEngine } from "./dspy-rlm-engine-DbTk4JdR.js";
|
|
21
21
|
import { f as Mutex, p as mapConcurrent } from "./ledger-core-DXZIqu17.js";
|
|
22
|
-
import { $ as
|
|
22
|
+
import { $ as REFERENCE_EQUIVALENCE_INPUT_LIMITS, B as buildReflectionPrompt, Ct as JudgeParseError, P as surfaceContentHash, St as summarizeBackendIntegrity, V as parseReflectionResponse, at as redTeamReport, bt as assertRealBackend, ct as Dataset, dt as llmJudge, et as REFERENCE_EQUIVALENCE_JUDGE_VERSION, ft as crowdingDistance, gt as scalarScore, ht as paretoFrontierWithCrowding, it as redTeamDataset, lt as HoldoutLockedError, mt as paretoFrontier, nt as runReferenceEquivalenceJudge, ot as scoreRedTeamOutput, pt as dominates, rt as DEFAULT_RED_TEAM_CORPUS, st as toolNamesForRun, tt as createReferenceEquivalenceJudge, ut as hashScenarios, vt as BackendIntegrityError, xt as summarizeAgentReceiptIntegrity, y as runCanaries, yt as assertRealAgentReceipts, z as DEFAULT_MUTATION_PRIMITIVES } from "./skillopt-optimization-method-CQdVeM8k.js";
|
|
23
23
|
import { $ as selfPreference, A as pairedRiskDifference, B as requiredSampleSize, C as mulberry32, D as pairedCohensDz, E as pairedBootstrap, F as partialCredit, G as wilson, H as weightedComposite, I as passAtK, J as normalCdf, K as studentTCdf, L as pearsonR, M as pairedRiskDifferenceScore, N as pairedSignTest, O as pairedDeltaTieFraction, P as pairedTTest, Q as positionalBias, R as ranks, S as mcnemarRequiredN, T as pairedBinaryScale, U as weightedMean, V as spearmanR, W as wilcoxonSignedRank, X as calibrateJudgeContinuous, Y as calibrateJudge, Z as continuousAgreement, _ as interpretCliffs, a as MANN_WHITNEY_EXACT_MAX_WORK, b as mcnemar, c as bonferroni, d as confidenceInterval, et as verbosityBias, f as corpusInterRaterAgreement, g as interRaterReliability, h as holm, i as MANN_WHITNEY_EXACT_MAX_STATES, j as pairedRiskDifferenceExact, k as pairedMde, l as cliffsDelta, m as eProcess, n as DECISION_PAIRED_DELTA_STATISTIC, o as WILCOXON_EXACT_MAX_N, p as corpusInterRaterAgreementFromJudgeScores, r as DEFAULT_PERMUTATIONS, s as benjaminiHochberg, t as BOOTSTRAP_GATE_MIN_N, u as cohensD, v as isBinaryOutcomeVector, w as normalizeScores, x as mcnemarPower, y as mannWhitneyU, z as requiredPairedSampleSize } from "./statistics-ByxzSiOM.js";
|
|
24
24
|
import { n as pairArms, r as pairRunRecords, t as comparePairedArms } from "./paired-arms-iZ08VFMN.js";
|
|
25
25
|
import { a as improvementVerdict, i as gitProvenanceReader, n as computeExperimentStats, o as inMemoryExperimentStore, r as fileExperimentStore, s as clusteredPairedBinary, t as ExperimentTracker } from "./experiment-tracker-CnRICnMl.js";
|
|
@@ -36,7 +36,7 @@ import { C as rollupSupervisorRuns, D as isUnavailable, E as SUPERVISOR_RUN_SCHE
|
|
|
36
36
|
import { A as TRACE_ANALYST_ACTOR_DESCRIPTION, C as domainEvidencePattern, D as tokenizeDomainWords, E as scoreTraceInsightReadiness, O as traceAnalystOnRunComplete, S as describeTraceInsightScope, T as planTraceInsightQuestions, _ as otlpToTraceRunRecords, a as convertTraceStoresToOtlp, b as buildTraceInsightPrompt, c as otelRunCompleteHook, d as captureFetchToRawSink, f as ToolTraceMissingError, g as otlpToRunRecords, h as otlpRowsToTraceRunRecords, i as iterateRawCalls, j as TRACE_ANALYST_ACTOR_DESCRIPTION_VERSION, k as analyzeTraces, l as OTEL_AGENT_EVAL_SCOPE, m as otlpRowsToRunRecords, n as ReplayCacheMissError, o as createOtelExporter, p as toolSpansToTraceAnalysisStore, r as createReplayFetch, s as createOtelTracingStore, t as ReplayCache, u as exportRunAsOtlp, v as flattenOtlpExportToNdjson, w as inferDomainKeywords, x as defaultTraceInsightPanel, y as buildTraceInsightContext } from "./replay-CohS93nE.js";
|
|
37
37
|
import { i as otlpTextToTraceAnalysisStore, n as OtlpFileTraceStore } from "./store-otlp-Dw8PPIlL.js";
|
|
38
38
|
import { n as extractUsageFromResponse, r as extractUsageFromSse, t as extractUsage } from "./extract-usage-CdZdoj1s.js";
|
|
39
|
-
import { B as harnessAxisOf, F as HARNESS_NATIVE_MODEL, G as parseCorrectnessResponse, H as completionVerdict, I as agentProfileHash, K as verifyCompletion, L as agentProfileId, P as CODING_HARNESSES, R as agentProfileModelId, U as createLlmCorrectnessChecker, V as extractProducedState, W as createTokenRecallChecker, z as expandProfileAxes } from "./campaign-
|
|
39
|
+
import { B as harnessAxisOf, F as HARNESS_NATIVE_MODEL, G as parseCorrectnessResponse, H as completionVerdict, I as agentProfileHash, K as verifyCompletion, L as agentProfileId, P as CODING_HARNESSES, R as agentProfileModelId, U as createLlmCorrectnessChecker, V as extractProducedState, W as createTokenRecallChecker, z as expandProfileAxes } from "./campaign-C2TTzQII.js";
|
|
40
40
|
import { n as iqr, r as welchsTTest, t as compareToBaseline } from "./baseline-C-GocmIW.js";
|
|
41
41
|
import { a as judgeSpans, c as runsForScenario, i as hasCapturedToolArgs, l as toolSpans, n as argHash, o as llmSpans, r as groupBy, s as runFailureClass, t as aggregateLlm } from "./query-Di7eEQ79.js";
|
|
42
42
|
import { a as checkBehavioralCanary, i as canaryLeakView, o as checkCanaries, r as HoldoutAuditor, s as runBehavioralCanaries, t as analyzeRuns } from "./analyze-runs-C30yljDJ.js";
|
|
@@ -48,7 +48,7 @@ import { i as redactValue, n as REDACTION_VERSION, r as redactString, t as DEFAU
|
|
|
48
48
|
import { n as DEFAULT_RULES, r as classifyFailure, t as computeToolUseMetrics } from "./tool-use-metrics-DEGMKycK.js";
|
|
49
49
|
import { t as analyzeSeries } from "./series-convergence-CjO2QdRW.js";
|
|
50
50
|
import { n as runCounterfactual, t as attributeCounterfactuals } from "./counterfactual-CWPTrMH7.js";
|
|
51
|
-
import { _ as deterministicSplit, g as BENCHMARK_SPLIT_SEED, t as benchmarks_exports } from "./benchmarks-
|
|
51
|
+
import { _ as deterministicSplit, g as BENCHMARK_SPLIT_SEED, t as benchmarks_exports } from "./benchmarks-Dzs8CKb1.js";
|
|
52
52
|
import { t as runEvalCampaign } from "./eval-campaign-B_7wcnav.js";
|
|
53
53
|
import { n as pairedEvalueSequence, t as evaluateInterimReleaseConfidence } from "./sequential-Br0mAPHA.js";
|
|
54
54
|
import { accessSync, appendFileSync, constants, cpSync, existsSync, mkdirSync, promises, readFileSync, readdirSync, statSync, writeFileSync } from "node:fs";
|
package/dist/openapi.json
CHANGED
|
@@ -2,7 +2,7 @@
|
|
|
2
2
|
"openapi": "3.1.0",
|
|
3
3
|
"info": {
|
|
4
4
|
"title": "@tangle-network/agent-eval — wire protocol",
|
|
5
|
-
"version": "0.144.
|
|
5
|
+
"version": "0.144.11",
|
|
6
6
|
"description": "HTTP and stdio RPC interface to agent-eval. The TypeScript runtime is the source of truth; this spec is the contract that cross-language clients (Python, Rust, Go) generate from.\n\nWire-protocol version: 1.0.0. Bumps on breaking changes to request/response schemas.",
|
|
7
7
|
"contact": {
|
|
8
8
|
"name": "Tangle Network",
|
package/dist/{skillopt-optimization-method-D6Q4dLbh.js → skillopt-optimization-method-CQdVeM8k.js}
RENAMED
|
@@ -1551,7 +1551,7 @@ function assertObservationShape(value, line) {
|
|
|
1551
1551
|
const observation = value;
|
|
1552
1552
|
if (!Number.isSafeInteger(observation.sequence) || Number(observation.sequence) <= 0) throw new Error(`external optimizer observation artifact line ${line} has an invalid sequence`);
|
|
1553
1553
|
if (observation.kind === "proposal") {
|
|
1554
|
-
assertExactKeys(observation, [
|
|
1554
|
+
assertExactKeys$1(observation, [
|
|
1555
1555
|
"candidate",
|
|
1556
1556
|
"candidateHash",
|
|
1557
1557
|
"kind",
|
|
@@ -1561,7 +1561,7 @@ function assertObservationShape(value, line) {
|
|
|
1561
1561
|
return;
|
|
1562
1562
|
}
|
|
1563
1563
|
if (observation.kind === "evaluation") {
|
|
1564
|
-
assertExactKeys(observation, [
|
|
1564
|
+
assertExactKeys$1(observation, [
|
|
1565
1565
|
"candidate",
|
|
1566
1566
|
"candidateHash",
|
|
1567
1567
|
"evaluationNumber",
|
|
@@ -1596,7 +1596,7 @@ function assertCandidateFields(observation, line) {
|
|
|
1596
1596
|
if (!isExternalTextCandidate(observation.candidate)) throw new Error(`external optimizer observation artifact line ${line} has an invalid candidate`);
|
|
1597
1597
|
if (typeof observation.candidateHash !== "string" || !/^[a-f0-9]{64}$/u.test(observation.candidateHash)) throw new Error(`external optimizer observation artifact line ${line} has an invalid candidateHash`);
|
|
1598
1598
|
}
|
|
1599
|
-
function assertExactKeys(value, expected, line) {
|
|
1599
|
+
function assertExactKeys$1(value, expected, line) {
|
|
1600
1600
|
assertAllowedKeys(value, expected, line);
|
|
1601
1601
|
for (const key of expected) if (!(key in value)) throw new Error(`external optimizer observation artifact line ${line} is missing ${key}`);
|
|
1602
1602
|
}
|
|
@@ -3163,6 +3163,191 @@ function assertExternalCostAccounting(value, calls, name) {
|
|
|
3163
3163
|
if (accounting.kind !== "external" || typeof accounting.reason !== "string" || !accounting.reason.trim() || accounting.reason.trim() !== accounting.reason) throw new Error(`${name}: optimizer returned invalid costAccounting`);
|
|
3164
3164
|
}
|
|
3165
3165
|
//#endregion
|
|
3166
|
+
//#region src/campaign/gepa-candidate-population.ts
|
|
3167
|
+
/**
|
|
3168
|
+
* Read GEPA's exact candidate graph from the artifact addressed by method provenance.
|
|
3169
|
+
*
|
|
3170
|
+
* The reader checks the supplied digest, declared byte count, run identity,
|
|
3171
|
+
* candidate surfaces, parent graph, selection scores, and configured bounds.
|
|
3172
|
+
* This proves that the bytes match the supplied summary. The caller remains
|
|
3173
|
+
* responsible for obtaining that summary from trusted method provenance.
|
|
3174
|
+
*/
|
|
3175
|
+
function readGepaCandidatePopulationArtifact(input) {
|
|
3176
|
+
assertGepaCandidatePopulationSummary(input.summary);
|
|
3177
|
+
const scenarioIds = scenarioIdSet(input.summary.scenarioIds);
|
|
3178
|
+
const storage = input.storage ?? fsCampaignStorage();
|
|
3179
|
+
const contents = storage.read(input.summary.path);
|
|
3180
|
+
if (contents === void 0) {
|
|
3181
|
+
const state = storage.exists(input.summary.path) ? "unreadable" : "missing";
|
|
3182
|
+
throw new Error(`GEPA candidate population artifact is ${state} at '${input.summary.path}'`);
|
|
3183
|
+
}
|
|
3184
|
+
const bytes = new TextEncoder().encode(contents).byteLength;
|
|
3185
|
+
if (bytes !== input.summary.bytes) throw new Error(`GEPA candidate population byte count mismatch at '${input.summary.path}': expected ${input.summary.bytes}, got ${bytes}`);
|
|
3186
|
+
if (BigInt(bytes) > maximumArtifactBytes(input.summary.maxCandidates, input.summary.maxCandidateChars, scenarioIds)) throw new Error(`GEPA candidate population exceeds its configured bounds at '${input.summary.path}'`);
|
|
3187
|
+
const sha256 = `sha256:${createHash("sha256").update(contents).digest("hex")}`;
|
|
3188
|
+
if (sha256 !== input.summary.sha256) throw new Error(`GEPA candidate population digest mismatch at '${input.summary.path}': expected ${input.summary.sha256}, got ${sha256}`);
|
|
3189
|
+
let raw;
|
|
3190
|
+
try {
|
|
3191
|
+
raw = JSON.parse(contents);
|
|
3192
|
+
} catch (cause) {
|
|
3193
|
+
throw new Error(`GEPA candidate population is not JSON at '${input.summary.path}'`, { cause });
|
|
3194
|
+
}
|
|
3195
|
+
assertExactKeys(raw, [
|
|
3196
|
+
"bestIndex",
|
|
3197
|
+
"candidates",
|
|
3198
|
+
"runId",
|
|
3199
|
+
"schemaVersion",
|
|
3200
|
+
"scope"
|
|
3201
|
+
], "artifact");
|
|
3202
|
+
if (raw.schemaVersion !== 1 || raw.scope !== "gepa-candidate-population") throw new Error("GEPA candidate population has an unsupported schema");
|
|
3203
|
+
if (raw.runId !== input.summary.runId) throw new Error("GEPA candidate population run ID differs from its summary");
|
|
3204
|
+
if (!Array.isArray(raw.candidates) || raw.candidates.length === 0) throw new Error("GEPA candidate population must contain candidates");
|
|
3205
|
+
if (raw.candidates.length !== input.summary.candidates || raw.candidates.length > input.summary.maxCandidates) throw new Error("GEPA candidate population count differs from its summary or configured bound");
|
|
3206
|
+
const bestIndex = raw.bestIndex;
|
|
3207
|
+
if (typeof bestIndex !== "number" || !Number.isSafeInteger(bestIndex) || bestIndex < 0 || bestIndex >= raw.candidates.length || bestIndex !== input.summary.bestIndex) throw new Error("GEPA candidate population has an invalid best index");
|
|
3208
|
+
const candidates = raw.candidates.map((candidate, index) => parseCandidate({
|
|
3209
|
+
candidate,
|
|
3210
|
+
index,
|
|
3211
|
+
maxCandidateChars: input.summary.maxCandidateChars,
|
|
3212
|
+
scenarioIds,
|
|
3213
|
+
expectsComponents: input.summary.surfaceKind === "components"
|
|
3214
|
+
}));
|
|
3215
|
+
return deepFreezeCanonicalJson({
|
|
3216
|
+
summary: structuredClone(input.summary),
|
|
3217
|
+
runId: raw.runId,
|
|
3218
|
+
bestIndex,
|
|
3219
|
+
candidates
|
|
3220
|
+
});
|
|
3221
|
+
}
|
|
3222
|
+
function assertGepaCandidatePopulationSummary(value) {
|
|
3223
|
+
assertExactKeys(value, [
|
|
3224
|
+
"bestIndex",
|
|
3225
|
+
"bytes",
|
|
3226
|
+
"candidates",
|
|
3227
|
+
"maxCandidateChars",
|
|
3228
|
+
"maxCandidates",
|
|
3229
|
+
"path",
|
|
3230
|
+
"runId",
|
|
3231
|
+
"scenarioIds",
|
|
3232
|
+
"scope",
|
|
3233
|
+
"sha256",
|
|
3234
|
+
"surfaceKind"
|
|
3235
|
+
], "summary");
|
|
3236
|
+
if (value.scope !== "gepa-candidate-population") throw new Error("GEPA candidate population summary has an invalid scope");
|
|
3237
|
+
if (typeof value.path !== "string" || value.path.trim().length === 0) throw new Error("GEPA candidate population summary has an invalid path");
|
|
3238
|
+
if (typeof value.sha256 !== "string" || !/^sha256:[a-f0-9]{64}$/u.test(value.sha256)) throw new Error("GEPA candidate population summary has an invalid SHA-256");
|
|
3239
|
+
if (typeof value.runId !== "string" || value.runId.trim().length === 0) throw new Error("GEPA candidate population summary has an invalid run ID");
|
|
3240
|
+
const bytes = value.bytes;
|
|
3241
|
+
const candidates = value.candidates;
|
|
3242
|
+
const maxCandidates = value.maxCandidates;
|
|
3243
|
+
const maxCandidateChars = value.maxCandidateChars;
|
|
3244
|
+
assertPositiveSafeInteger$2(bytes, "summary.bytes");
|
|
3245
|
+
assertPositiveSafeInteger$2(candidates, "summary.candidates");
|
|
3246
|
+
assertPositiveSafeInteger$2(maxCandidates, "summary.maxCandidates");
|
|
3247
|
+
assertPositiveSafeInteger$2(maxCandidateChars, "summary.maxCandidateChars");
|
|
3248
|
+
if (candidates > maxCandidates) throw new Error("GEPA candidate population summary exceeds its configured candidate bound");
|
|
3249
|
+
scenarioIdSet(value.scenarioIds);
|
|
3250
|
+
if (value.surfaceKind !== "text" && value.surfaceKind !== "components") throw new Error("GEPA candidate population summary has an invalid surface kind");
|
|
3251
|
+
const bestIndex = value.bestIndex;
|
|
3252
|
+
if (typeof bestIndex !== "number" || !Number.isSafeInteger(bestIndex) || bestIndex < 0 || bestIndex >= candidates) throw new Error("GEPA candidate population summary has an invalid best index");
|
|
3253
|
+
}
|
|
3254
|
+
function parseCandidate(input) {
|
|
3255
|
+
const { candidate, index } = input;
|
|
3256
|
+
assertExactKeys(candidate, [
|
|
3257
|
+
"aggregateScore",
|
|
3258
|
+
"candidate",
|
|
3259
|
+
"discoveryEvaluationCount",
|
|
3260
|
+
"index",
|
|
3261
|
+
"parentIndices",
|
|
3262
|
+
"selectionScores"
|
|
3263
|
+
], `candidate ${index}`);
|
|
3264
|
+
if (candidate.index !== index) throw new Error(`GEPA candidate population expected index ${index}`);
|
|
3265
|
+
if (!isExternalTextCandidate(candidate.candidate) || input.expectsComponents !== (typeof candidate.candidate !== "string") || candidateChars(candidate.candidate) > input.maxCandidateChars) throw new Error(`GEPA candidate ${index} has an invalid surface`);
|
|
3266
|
+
const parentIndices = parseParents(candidate.parentIndices, index);
|
|
3267
|
+
const selectionScores = parseSelectionScores(candidate.selectionScores, input.scenarioIds, index);
|
|
3268
|
+
const aggregateScore = parseAggregateScore(candidate.aggregateScore, selectionScores, index);
|
|
3269
|
+
const discoveryEvaluationCount = candidate.discoveryEvaluationCount;
|
|
3270
|
+
if (typeof discoveryEvaluationCount !== "number" || !Number.isSafeInteger(discoveryEvaluationCount) || discoveryEvaluationCount < 0) throw new Error(`GEPA candidate ${index} has an invalid discovery evaluation count`);
|
|
3271
|
+
const candidateHash = contentHash({
|
|
3272
|
+
kind: "external-text-candidate",
|
|
3273
|
+
candidate: candidate.candidate
|
|
3274
|
+
});
|
|
3275
|
+
return {
|
|
3276
|
+
index,
|
|
3277
|
+
candidate: structuredClone(candidate.candidate),
|
|
3278
|
+
candidateHash,
|
|
3279
|
+
candidateDigest: `sha256:${candidateHash}`,
|
|
3280
|
+
parentIndices,
|
|
3281
|
+
aggregateScore,
|
|
3282
|
+
selectionScores,
|
|
3283
|
+
discoveryEvaluationCount
|
|
3284
|
+
};
|
|
3285
|
+
}
|
|
3286
|
+
function parseParents(value, index) {
|
|
3287
|
+
if (!Array.isArray(value) || value.length === 0) throw new Error(`GEPA candidate ${index} has invalid parents`);
|
|
3288
|
+
const seen = /* @__PURE__ */ new Set();
|
|
3289
|
+
for (const parent of value) {
|
|
3290
|
+
if (parent === null) {
|
|
3291
|
+
if (index !== 0) throw new Error(`GEPA candidate ${index} has a null parent`);
|
|
3292
|
+
} else if (!Number.isSafeInteger(parent) || parent < 0 || parent >= index) throw new Error(`GEPA candidate ${index} has an invalid parent index`);
|
|
3293
|
+
if (seen.has(parent)) throw new Error(`GEPA candidate ${index} repeats a parent index`);
|
|
3294
|
+
seen.add(parent);
|
|
3295
|
+
}
|
|
3296
|
+
if (index === 0 && (value.length !== 1 || value[0] !== null)) throw new Error("GEPA root candidate must have one null parent");
|
|
3297
|
+
return [...value];
|
|
3298
|
+
}
|
|
3299
|
+
function parseSelectionScores(value, scenarioIds, candidateIndex) {
|
|
3300
|
+
if (!Array.isArray(value)) throw new Error(`GEPA candidate ${candidateIndex} has invalid selection scores`);
|
|
3301
|
+
const seen = /* @__PURE__ */ new Set();
|
|
3302
|
+
let prior = "";
|
|
3303
|
+
return value.map((row, index) => {
|
|
3304
|
+
assertExactKeys(row, ["scenarioId", "score"], `candidate ${candidateIndex} score ${index}`);
|
|
3305
|
+
if (typeof row.scenarioId !== "string" || !scenarioIds.has(row.scenarioId) || seen.has(row.scenarioId) || index > 0 && row.scenarioId <= prior) throw new Error(`GEPA candidate ${candidateIndex} has invalid selection score identity`);
|
|
3306
|
+
if (typeof row.score !== "number" || !Number.isFinite(row.score)) throw new Error(`GEPA candidate ${candidateIndex} has an invalid selection score`);
|
|
3307
|
+
seen.add(row.scenarioId);
|
|
3308
|
+
prior = row.scenarioId;
|
|
3309
|
+
return {
|
|
3310
|
+
scenarioId: row.scenarioId,
|
|
3311
|
+
score: row.score
|
|
3312
|
+
};
|
|
3313
|
+
});
|
|
3314
|
+
}
|
|
3315
|
+
function parseAggregateScore(value, scores, candidateIndex) {
|
|
3316
|
+
if (scores.length === 0) {
|
|
3317
|
+
if (value !== null) throw new Error(`GEPA candidate ${candidateIndex} has a score without selection evidence`);
|
|
3318
|
+
return null;
|
|
3319
|
+
}
|
|
3320
|
+
if (typeof value !== "number" || !Number.isFinite(value)) throw new Error(`GEPA candidate ${candidateIndex} has an invalid aggregate score`);
|
|
3321
|
+
const mean = scores.reduce((total, row) => total + row.score, 0) / scores.length;
|
|
3322
|
+
if (Math.abs(value - mean) > Math.max(1e-12, Math.abs(mean) * 1e-9)) throw new Error(`GEPA candidate ${candidateIndex} aggregate score differs from its selection scores`);
|
|
3323
|
+
return value;
|
|
3324
|
+
}
|
|
3325
|
+
function candidateChars(candidate) {
|
|
3326
|
+
return typeof candidate === "string" ? candidate.length : JSON.stringify(candidate).length;
|
|
3327
|
+
}
|
|
3328
|
+
function scenarioIdSet(values) {
|
|
3329
|
+
if (!Array.isArray(values)) throw new Error("scenarioIds must be an array");
|
|
3330
|
+
const result = /* @__PURE__ */ new Set();
|
|
3331
|
+
for (const value of values) {
|
|
3332
|
+
if (typeof value !== "string" || value.length === 0 || result.has(value)) throw new Error("scenarioIds must contain unique non-empty strings");
|
|
3333
|
+
result.add(value);
|
|
3334
|
+
}
|
|
3335
|
+
return result;
|
|
3336
|
+
}
|
|
3337
|
+
function maximumArtifactBytes(maxCandidates, maxCandidateChars, scenarioIds) {
|
|
3338
|
+
const scenarioChars = [...scenarioIds].reduce((total, id) => total + id.length, 0);
|
|
3339
|
+
const candidates = BigInt(maxCandidates);
|
|
3340
|
+
return 8192n + candidates * (BigInt(maxCandidateChars) * 6n + BigInt(scenarioChars) * 6n + BigInt(scenarioIds.size) * 128n + candidates * 32n + 8192n);
|
|
3341
|
+
}
|
|
3342
|
+
function assertExactKeys(value, expected, label) {
|
|
3343
|
+
if (!isRecord(value)) throw new Error(`GEPA candidate population ${label} must be an object`);
|
|
3344
|
+
const actual = Object.keys(value).sort();
|
|
3345
|
+
if (actual.length !== expected.length || actual.some((key, index) => key !== expected[index])) throw new Error(`GEPA candidate population ${label} has unexpected fields`);
|
|
3346
|
+
}
|
|
3347
|
+
function assertPositiveSafeInteger$2(value, label) {
|
|
3348
|
+
if (!Number.isSafeInteger(value) || value <= 0) throw new Error(`GEPA candidate population ${label} must be a positive safe integer`);
|
|
3349
|
+
}
|
|
3350
|
+
//#endregion
|
|
3166
3351
|
//#region src/campaign/presets/compare-optimization-methods.ts
|
|
3167
3352
|
/**
|
|
3168
3353
|
* Compare optimization methods on shared train, selection, and test data.
|
|
@@ -3417,6 +3602,14 @@ function assertOptimizationProvenance(methodName, value) {
|
|
|
3417
3602
|
"refusals"
|
|
3418
3603
|
]) if (!Number.isSafeInteger(value.observations[field]) || value.observations[field] < 0) fail(`observations.${field}`);
|
|
3419
3604
|
}
|
|
3605
|
+
if (value.gepaCandidatePopulation !== void 0) {
|
|
3606
|
+
try {
|
|
3607
|
+
assertGepaCandidatePopulationSummary(value.gepaCandidatePopulation);
|
|
3608
|
+
} catch {
|
|
3609
|
+
fail("gepaCandidatePopulation");
|
|
3610
|
+
}
|
|
3611
|
+
if (value.gepaCandidatePopulation.runId !== value.runId) fail("gepaCandidatePopulation.runId");
|
|
3612
|
+
}
|
|
3420
3613
|
if (value.modelExecutions !== void 0) {
|
|
3421
3614
|
if (value.modelExecutions.scope !== "runtime-model-calls" || typeof value.modelExecutions.path !== "string" || !value.modelExecutions.path.trim() || typeof value.modelExecutions.sha256 !== "string" || !/^sha256:[0-9a-f]{64}$/.test(value.modelExecutions.sha256)) fail("modelExecutions");
|
|
3422
3615
|
for (const field of [
|
|
@@ -4994,7 +5187,7 @@ function addEvaluationLimit(total, increment) {
|
|
|
4994
5187
|
}
|
|
4995
5188
|
//#endregion
|
|
4996
5189
|
//#region src/campaign/gepa-optimization-result.ts
|
|
4997
|
-
function assertGepaBridgeOutput(result, name, maxCandidateChars, recipeKind, maxEvaluations, expectsComponents) {
|
|
5190
|
+
function assertGepaBridgeOutput(result, name, maxCandidateChars, recipeKind, maxEvaluations, maxPopulationCandidates, scenarioIds, expectsComponents, requiresCandidatePopulation) {
|
|
4998
5191
|
if (result.recipeKind !== recipeKind) throw new Error(`${name}: GEPA bridge reported recipe '${String(result.recipeKind)}'`);
|
|
4999
5192
|
if (!isGepaCandidate(result.bestCandidate, maxCandidateChars) || expectsComponents !== (typeof result.bestCandidate !== "string")) throw new Error(`${name}: GEPA bridge returned an invalid candidate`);
|
|
5000
5193
|
if (!Number.isFinite(result.bestScore)) throw new Error(`${name}: GEPA bridge returned an invalid bestScore`);
|
|
@@ -5007,6 +5200,11 @@ function assertGepaBridgeOutput(result, name, maxCandidateChars, recipeKind, max
|
|
|
5007
5200
|
assertExternalOptimizerPackageSource(result.upstream, "gepa", name, "GEPA");
|
|
5008
5201
|
if (typeof result.runId !== "string" || result.runId.length === 0 || result.runId !== result.runId.trim()) throw new Error(`${name}: GEPA bridge returned an invalid runId`);
|
|
5009
5202
|
if (typeof result.resumed !== "boolean") throw new Error(`${name}: GEPA bridge returned an invalid resumed flag`);
|
|
5203
|
+
if (result.candidatePopulation !== void 0) {
|
|
5204
|
+
assertGepaCandidatePopulationSummary(result.candidatePopulation);
|
|
5205
|
+
if (result.candidatePopulation.runId !== result.runId) throw new Error(`${name}: GEPA candidate population has a different run ID`);
|
|
5206
|
+
if (result.candidatePopulation.maxCandidates !== maxPopulationCandidates || result.candidatePopulation.maxCandidateChars !== maxCandidateChars || result.candidatePopulation.surfaceKind !== (expectsComponents ? "components" : "text") || result.candidatePopulation.scenarioIds.length !== scenarioIds.length || result.candidatePopulation.scenarioIds.some((scenarioId, index) => scenarioId !== scenarioIds[index])) throw new Error(`${name}: GEPA candidate population differs from its configured bounds`);
|
|
5207
|
+
} else if (requiresCandidatePopulation) throw new Error(`${name}: GEPA bridge omitted the official candidate population`);
|
|
5010
5208
|
}
|
|
5011
5209
|
function isGepaCandidate(value, maxChars) {
|
|
5012
5210
|
if (!isExternalTextCandidate(value)) return false;
|
|
@@ -5060,6 +5258,8 @@ function gepaOptimizationMethod(config) {
|
|
|
5060
5258
|
const trainSet = input.trainScenarios.map((scenario) => describeExternalScenario(scenario, "GEPA", maxEvidenceChars, config.describeScenario));
|
|
5061
5259
|
const selectionSet = input.selectionScenarios.map((scenario) => describeExternalScenario(scenario, "GEPA", maxEvidenceChars, config.describeScenario));
|
|
5062
5260
|
const evaluationLimit = gepaRecipeEvaluationLimit(config.recipe, input.selectionScenarios.length);
|
|
5261
|
+
const maxPopulationCandidates = Math.min(Number.MAX_SAFE_INTEGER, evaluationLimit + 1);
|
|
5262
|
+
const populationScenarioIds = (selectionSet.length > 0 ? selectionSet : trainSet).map((scenario) => scenario.id);
|
|
5063
5263
|
const runMaterial = {
|
|
5064
5264
|
optimizer: "gepa",
|
|
5065
5265
|
runtime: runtimeIdentity,
|
|
@@ -5075,6 +5275,7 @@ function gepaOptimizationMethod(config) {
|
|
|
5075
5275
|
trainSet,
|
|
5076
5276
|
selectionSet,
|
|
5077
5277
|
maxCandidateChars,
|
|
5278
|
+
maxPopulationCandidates,
|
|
5078
5279
|
maxEvidenceChars,
|
|
5079
5280
|
evaluationCallbackLimits: resolveExternalOptimizerCallbackLimits(config.evaluationCallbackLimits),
|
|
5080
5281
|
optimizerModel: config.optimizer ? {
|
|
@@ -5188,6 +5389,7 @@ function gepaOptimizationMethod(config) {
|
|
|
5188
5389
|
trainSet,
|
|
5189
5390
|
selectionSet,
|
|
5190
5391
|
maxCandidateChars,
|
|
5392
|
+
maxPopulationCandidates,
|
|
5191
5393
|
maxEvidenceChars,
|
|
5192
5394
|
outputDir,
|
|
5193
5395
|
...modelProxy && config.optimizer ? { modelProxy: {
|
|
@@ -5210,7 +5412,7 @@ function gepaOptimizationMethod(config) {
|
|
|
5210
5412
|
cleanup: closeResources
|
|
5211
5413
|
});
|
|
5212
5414
|
signal?.throwIfAborted();
|
|
5213
|
-
assertGepaBridgeOutput(result, name, maxCandidateChars, config.recipe.kind, evaluationLimit, expectsComponents);
|
|
5415
|
+
assertGepaBridgeOutput(result, name, maxCandidateChars, config.recipe.kind, evaluationLimit, maxPopulationCandidates, populationScenarioIds, expectsComponents, config.recipe.kind === "engine" && config.recipe.run.engine === "gepa");
|
|
5214
5416
|
assertExternalOptimizerRunBinding({
|
|
5215
5417
|
label: name,
|
|
5216
5418
|
runtime: runtimeIdentity,
|
|
@@ -5222,6 +5424,15 @@ function gepaOptimizationMethod(config) {
|
|
|
5222
5424
|
resumed: result.resumed
|
|
5223
5425
|
});
|
|
5224
5426
|
if (callback.evaluations() !== result.totalEvaluations) throw new Error(`${name}: GEPA reported ${result.totalEvaluations} evaluations but the callback received ${callback.evaluations()}`);
|
|
5427
|
+
if (result.candidatePopulation) {
|
|
5428
|
+
const population = readGepaCandidatePopulationArtifact({ summary: result.candidatePopulation });
|
|
5429
|
+
const selected = population.candidates[population.bestIndex];
|
|
5430
|
+
const selectedHash = contentHash({
|
|
5431
|
+
kind: "external-text-candidate",
|
|
5432
|
+
candidate: result.bestCandidate
|
|
5433
|
+
});
|
|
5434
|
+
if (selected?.candidateHash !== selectedHash) throw new Error(`${name}: GEPA candidate population identifies a different winner`);
|
|
5435
|
+
}
|
|
5225
5436
|
const evaluationCost = costFromLedgerSummary(costLedger.summary({
|
|
5226
5437
|
phase: "gepa.external-evaluation",
|
|
5227
5438
|
tags: runBudget.runTags
|
|
@@ -5282,6 +5493,7 @@ function gepaOptimizationMethod(config) {
|
|
|
5282
5493
|
artifactDir: outputDir,
|
|
5283
5494
|
...tokenUsage ? { tokenUsage } : {},
|
|
5284
5495
|
observations: observationLog.summary(),
|
|
5496
|
+
...result.candidatePopulation ? { gepaCandidatePopulation: result.candidatePopulation } : {},
|
|
5285
5497
|
...executionLog ? { modelExecutions: executionLog.summary() } : {}
|
|
5286
5498
|
}
|
|
5287
5499
|
};
|
|
@@ -6695,6 +6907,6 @@ function skillOptOptimizationMethod(config) {
|
|
|
6695
6907
|
};
|
|
6696
6908
|
}
|
|
6697
6909
|
//#endregion
|
|
6698
|
-
export {
|
|
6910
|
+
export { REFERENCE_EQUIVALENCE_INPUT_LIMITS as $, assertComponentSurface as A, buildReflectionPrompt as B, combineComparisonCosts as C, JudgeParseError as Ct, readGepaCandidatePopulationArtifact as D, optimizationTokenUsageFromSummary as E, surfaceHash as F, tangleTracesRoot as G, planCampaignRun as H, campaignBreakdown as I, assertCampaignSplitIdentity as J, readExternalOptimizerObservationArtifact as K, campaignMeanComposite as L, componentSurfaceIdentityMaterial as M, renderSurfaceDiff as N, decodeExternalTextCandidate as O, surfaceContentHash as P, openAutoPr as Q, compareRankKeys as R, assertOptimizationResult as S, summarizeBackendIntegrity as St, costFromLedgerSummary as T, runCampaign as U, parseReflectionResponse as V, resolveRunDir as W, campaignSplitDigest as X, campaignScenarioIdentity as Y, campaignSplitDigestFromIdentities as Z, heldOutGate as _, recoverTruncatedJson as _t, emitLoopProvenance as a, redTeamReport as at, composeGate as b, assertRealBackend as bt, provenanceRecordPath as c, Dataset as ct, runImprovementLoop as d, llmJudge as dt, REFERENCE_EQUIVALENCE_JUDGE_VERSION as et, runOptimization as f, crowdingDistance as ft, gepaOptimizationMethod as g, scalarScore as gt, labelTrustRank as h, paretoFrontierWithCrowding as ht, canonicalDigest as i, redTeamDataset as it, codeSurfaceIdentityMaterial as j, assertCodeSurfaceIdentity as k, provenanceSpansPath as l, HoldoutLockedError as lt, isProposedCandidate as m, paretoFrontier as mt, buildLoopProvenanceRecord as n, runReferenceEquivalenceJudge as nt, loopProvenanceArgsFromResult as o, scoreRedTeamOutput as ot, runEval as p, dominates as pt, assertCampaignDesign as q, campaignMeasurementDigest as r, DEFAULT_RED_TEAM_CORPUS as rt, loopProvenanceSpans as s, toolNamesForRun as st, skillOptOptimizationMethod as t, createReferenceEquivalenceJudge as tt, verifyLoopProvenanceRecord as u, hashScenarios as ut, defaultProductionGate as v, BackendIntegrityError as vt, compareOptimizationMethods as w, externalTextOptimizationMethod as x, summarizeAgentReceiptIntegrity as xt, runCanaries as y, assertRealAgentReceipts as yt, DEFAULT_MUTATION_PRIMITIVES as z };
|
|
6699
6911
|
|
|
6700
|
-
//# sourceMappingURL=skillopt-optimization-method-
|
|
6912
|
+
//# sourceMappingURL=skillopt-optimization-method-CQdVeM8k.js.map
|