@tangle-network/agent-eval 0.145.8 → 0.145.10
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +17 -0
- package/dist/analyst/index.d.ts +1 -1
- package/dist/analyst/index.js +1 -1
- package/dist/{benchmark-command-rWP4-dJ6.js → benchmark-command-BDHW2AUg.js} +2 -2
- package/dist/{benchmark-command-rWP4-dJ6.js.map → benchmark-command-BDHW2AUg.js.map} +1 -1
- package/dist/benchmarks/index.js +1 -1
- package/dist/campaign/index.js +1 -1
- package/dist/{campaign-Dg2B4W6d.js → campaign-B3GJpJ4h.js} +6 -3
- package/dist/campaign-B3GJpJ4h.js.map +1 -0
- package/dist/cli.js +1 -1
- package/dist/index-CM-e2LiV.d.ts.map +1 -1
- package/dist/openapi.json +1 -1
- package/package.json +3 -3
- package/dist/campaign-Dg2B4W6d.js.map +0 -1
package/dist/benchmarks/index.js
CHANGED
|
@@ -1,7 +1,7 @@
|
|
|
1
1
|
import { t as __exportAll } from "../rolldown-runtime-8H4AJuhK.js";
|
|
2
2
|
import { K as runCampaign } from "../llm-judge-DhiJqSjB.js";
|
|
3
3
|
import { x as fsCampaignStorage } from "../external-optimizer-subprocess-BhKYK0Jv.js";
|
|
4
|
-
import "../campaign-
|
|
4
|
+
import "../campaign-B3GJpJ4h.js";
|
|
5
5
|
import { join } from "node:path";
|
|
6
6
|
//#region src/benchmarks/calibration.ts
|
|
7
7
|
async function calibrateBenchmarkMetric(options) {
|
package/dist/campaign/index.js
CHANGED
|
@@ -4,7 +4,7 @@ import { a as heldoutSignificance, i as dimensionRegressions, o as pairHoldout,
|
|
|
4
4
|
import { r as makeProposalFinding } from "../types-BI4fT3HN.js";
|
|
5
5
|
import { n as acquireSingleRunLock } from "../external-optimizer-process-BRE56woM.js";
|
|
6
6
|
import { a as externalTextOptimizationMethod, i as composeGate, n as gepaOptimizationMethod, o as decodeExternalTextCandidate, r as heldOutGate, s as readExternalOptimizerObservationArtifact, t as skillOptOptimizationMethod, u as createReferenceEquivalenceJudge } from "../skillopt-optimization-method-C6JSfYxT.js";
|
|
7
|
-
import { A as neutralizationGate, C as scoreboardSummary, D as LabeledScenarioStoreError, E as FsLabeledScenarioStore, F as analyzeCrossSurfaceInteractions, I as buildTraceAnalystSurfaceDispatch, L as traceAnalystQualityJudge, M as loadEvalFixture, N as loadEvalFixtureScenarios, O as classifyUngroundedLiterals, P as planEvalFixtureRun, S as scoreUserStory, T as neutralizeText, _ as runProfileMatrixSegment, a as autoevalsScorerJudge, b as makePlaybackDispatch, c as FileSearchLedger, d as validateSearchLedgerEvent, f as scoreDiscrimination, g as finalizeProfileMatrix, h as createProfileMatrixPlan, i as verifyCodeSurface, j as discoverEvalFixtures, k as rolloutArgumentDiff, l as SEARCH_LEDGER_SCHEMA, m as SegmentedProfileMatrixError, n as gitWorktreeAdapter, o as phoenixEvaluatorJudge, p as selectDiscriminative, r as resolveWorktreePath, s as isTransientTransportFailure, t as WorktreeAdapterError, u as openSearchLedger, v as ProfileMatrixError, w as userStoryScoreboard, x as renderScoreboardMarkdown, y as runProfileMatrix } from "../campaign-
|
|
7
|
+
import { A as neutralizationGate, C as scoreboardSummary, D as LabeledScenarioStoreError, E as FsLabeledScenarioStore, F as analyzeCrossSurfaceInteractions, I as buildTraceAnalystSurfaceDispatch, L as traceAnalystQualityJudge, M as loadEvalFixture, N as loadEvalFixtureScenarios, O as classifyUngroundedLiterals, P as planEvalFixtureRun, S as scoreUserStory, T as neutralizeText, _ as runProfileMatrixSegment, a as autoevalsScorerJudge, b as makePlaybackDispatch, c as FileSearchLedger, d as validateSearchLedgerEvent, f as scoreDiscrimination, g as finalizeProfileMatrix, h as createProfileMatrixPlan, i as verifyCodeSurface, j as discoverEvalFixtures, k as rolloutArgumentDiff, l as SEARCH_LEDGER_SCHEMA, m as SegmentedProfileMatrixError, n as gitWorktreeAdapter, o as phoenixEvaluatorJudge, p as selectDiscriminative, r as resolveWorktreePath, s as isTransientTransportFailure, t as WorktreeAdapterError, u as openSearchLedger, v as ProfileMatrixError, w as userStoryScoreboard, x as renderScoreboardMarkdown, y as runProfileMatrix } from "../campaign-B3GJpJ4h.js";
|
|
8
8
|
import { n as paretoPolicy, r as paretoSignificanceGate, t as buildEvidenceVector } from "../promotion-policy-xzA40Evo.js";
|
|
9
9
|
import { n as sequentialPairedGate, t as sequentialDecide } from "../sequential-C458DXNf.js";
|
|
10
10
|
export { DEFAULT_EXTERNAL_OPTIMIZER_CALLBACK_LIMITS, DEFAULT_EXTERNAL_OPTIMIZER_PROCESS_LIMITS, FileSearchLedger, FsLabeledScenarioStore, LabeledScenarioStoreError, ProfileMatrixError, SEARCH_LEDGER_SCHEMA, SearchLedgerConflictError, SearchLedgerError, SearchLedgerIntegrityError, SegmentedProfileMatrixError, WorktreeAdapterError, acquireSingleRunLock, analyzeCrossSurfaceInteractions, assertCampaignDesign, assertCampaignSplitIdentity, assertCodeSurfaceIdentity, assertComponentSurface, autoevalsScorerJudge, buildEvidenceVector, buildLoopProvenanceRecord, buildTraceAnalystSurfaceDispatch, campaignBreakdown, campaignMeanComposite, campaignMeasurementDigest, campaignScenarioIdentity, campaignSplitDigest, campaignSplitDigestFromIdentities, canonicalDigest, cellCachePath, classifyUngroundedLiterals, codeSurfaceIdentityMaterial, combineComparisonCosts, compareOptimizationMethods, compareRankKeys, componentSurfaceIdentityMaterial, composeGate, costFromLedgerSummary, createProfileMatrixPlan, createReferenceEquivalenceJudge, createRunCostLedger, decodeExternalTextCandidate, defaultProductionGate, detectScale, dimensionRegressions, discoverEvalFixtures, emitLoopProvenance, externalTextOptimizationMethod, finalizeProfileMatrix, fsCampaignStorage, gepaOptimizationMethod, gitWorktreeAdapter, heldOutGate, heldoutSignificance, inMemoryCampaignStorage, isProposedCandidate, isTransientTransportFailure, labelTrustRank, llmJudge, loadEvalFixture, loadEvalFixtureScenarios, loopProvenanceArgsFromResult, loopProvenanceSpans, makePlaybackDispatch, makeProposalFinding, neutralizationGate, neutralizeText, openAutoPr, openSearchLedger, optimizationTokenUsageFromSummary, pairHoldout, paretoPolicy, paretoSignificanceGate, phoenixEvaluatorJudge, planCampaignRun, planEvalFixtureRun, powerPreflight, provenanceRecordPath, provenanceSpansPath, readCachedCell, readExternalOptimizerObservationArtifact, readGepaCandidatePopulationArtifact, renderScoreboardMarkdown, renderSurfaceDiff, resolveExternalOptimizerCallbackLimits, resolveExternalOptimizerProcessLimits, resolveRunDir, resolveWorktreePath, rolloutArgumentDiff, runCampaign, runEval, runImprovementLoop, runOptimization, runProfileMatrix, runProfileMatrixSegment, scoreDiscrimination, scoreUserStory, scoreboardSummary, selectDiscriminative, sequentialDecide, sequentialPairedGate, skillOptOptimizationMethod, surfaceContentHash, surfaceHash, tangleTracesRoot, traceAnalystQualityJudge, userStoryScoreboard, validateSearchLedgerEvent, verifyCodeSurface, verifyLoopProvenanceRecord };
|
|
@@ -1542,10 +1542,13 @@ function receiptModels(cell) {
|
|
|
1542
1542
|
const reported = cell.resolvedModels ?? (cell.resolvedModel ? [cell.resolvedModel] : []);
|
|
1543
1543
|
return [...new Set(reported.map((model) => model.trim()).filter(Boolean))];
|
|
1544
1544
|
}
|
|
1545
|
+
function isUnknownModelOnly(reported) {
|
|
1546
|
+
return reported.length > 0 && reported.every((model) => model === "unknown");
|
|
1547
|
+
}
|
|
1545
1548
|
/** Resolve and validate every paid-call model used by one cell. */
|
|
1546
1549
|
function recordModel(cell, profileId, declaredModel) {
|
|
1547
1550
|
const reported = receiptModels(cell);
|
|
1548
|
-
if (
|
|
1551
|
+
if (cell.error !== void 0 && (reported.length === 0 || isUnknownModelOnly(reported))) return UNKNOWN_MODEL;
|
|
1549
1552
|
if (modelHasSnapshot(declaredModel)) {
|
|
1550
1553
|
for (const model of reported) if (model !== declaredModel) throw new ProfileMatrixError(`profile '${profileId}' paid-call model '${model}' for cell '${cell.cellId}' does not match its declared exact model '${declaredModel}'`);
|
|
1551
1554
|
return declaredModel;
|
|
@@ -1595,7 +1598,7 @@ function buildRunRecord(args) {
|
|
|
1595
1598
|
*/
|
|
1596
1599
|
function durableCellForRecord(cell) {
|
|
1597
1600
|
if (cell.error === void 0) return cell;
|
|
1598
|
-
const hasSettledAgentCalls = receiptModels(cell).
|
|
1601
|
+
const hasSettledAgentCalls = receiptModels(cell).some((model) => model !== UNKNOWN_MODEL);
|
|
1599
1602
|
return {
|
|
1600
1603
|
...cell,
|
|
1601
1604
|
tokenUsage: cell.tokenUsage.tokensKnown === false || hasSettledAgentCalls ? cell.tokenUsage : {
|
|
@@ -3682,4 +3685,4 @@ function resolveWorktreePath(surface, worktreeDir) {
|
|
|
3682
3685
|
//#endregion
|
|
3683
3686
|
export { neutralizationGate as A, scoreboardSummary as C, LabeledScenarioStoreError as D, FsLabeledScenarioStore as E, analyzeCrossSurfaceInteractions as F, buildTraceAnalystSurfaceDispatch as I, traceAnalystQualityJudge as L, loadEvalFixture as M, loadEvalFixtureScenarios as N, classifyUngroundedLiterals as O, planEvalFixtureRun as P, scoreUserStory as S, neutralizeText as T, runProfileMatrixSegment as _, autoevalsScorerJudge as a, makePlaybackDispatch as b, FileSearchLedger as c, validateSearchLedgerEvent as d, scoreDiscrimination as f, finalizeProfileMatrix as g, createProfileMatrixPlan as h, verifyCodeSurface as i, discoverEvalFixtures as j, rolloutArgumentDiff as k, SEARCH_LEDGER_SCHEMA as l, SegmentedProfileMatrixError as m, gitWorktreeAdapter as n, phoenixEvaluatorJudge as o, selectDiscriminative as p, resolveWorktreePath as r, isTransientTransportFailure as s, WorktreeAdapterError as t, openSearchLedger as u, ProfileMatrixError as v, userStoryScoreboard as w, renderScoreboardMarkdown as x, runProfileMatrix as y };
|
|
3684
3687
|
|
|
3685
|
-
//# sourceMappingURL=campaign-
|
|
3688
|
+
//# sourceMappingURL=campaign-B3GJpJ4h.js.map
|