@tangle-network/agent-eval 0.145.8 → 0.145.10

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -1,7 +1,7 @@
1
1
  import { t as __exportAll } from "../rolldown-runtime-8H4AJuhK.js";
2
2
  import { K as runCampaign } from "../llm-judge-DhiJqSjB.js";
3
3
  import { x as fsCampaignStorage } from "../external-optimizer-subprocess-BhKYK0Jv.js";
4
- import "../campaign-Dg2B4W6d.js";
4
+ import "../campaign-B3GJpJ4h.js";
5
5
  import { join } from "node:path";
6
6
  //#region src/benchmarks/calibration.ts
7
7
  async function calibrateBenchmarkMetric(options) {
@@ -4,7 +4,7 @@ import { a as heldoutSignificance, i as dimensionRegressions, o as pairHoldout,
4
4
  import { r as makeProposalFinding } from "../types-BI4fT3HN.js";
5
5
  import { n as acquireSingleRunLock } from "../external-optimizer-process-BRE56woM.js";
6
6
  import { a as externalTextOptimizationMethod, i as composeGate, n as gepaOptimizationMethod, o as decodeExternalTextCandidate, r as heldOutGate, s as readExternalOptimizerObservationArtifact, t as skillOptOptimizationMethod, u as createReferenceEquivalenceJudge } from "../skillopt-optimization-method-C6JSfYxT.js";
7
- import { A as neutralizationGate, C as scoreboardSummary, D as LabeledScenarioStoreError, E as FsLabeledScenarioStore, F as analyzeCrossSurfaceInteractions, I as buildTraceAnalystSurfaceDispatch, L as traceAnalystQualityJudge, M as loadEvalFixture, N as loadEvalFixtureScenarios, O as classifyUngroundedLiterals, P as planEvalFixtureRun, S as scoreUserStory, T as neutralizeText, _ as runProfileMatrixSegment, a as autoevalsScorerJudge, b as makePlaybackDispatch, c as FileSearchLedger, d as validateSearchLedgerEvent, f as scoreDiscrimination, g as finalizeProfileMatrix, h as createProfileMatrixPlan, i as verifyCodeSurface, j as discoverEvalFixtures, k as rolloutArgumentDiff, l as SEARCH_LEDGER_SCHEMA, m as SegmentedProfileMatrixError, n as gitWorktreeAdapter, o as phoenixEvaluatorJudge, p as selectDiscriminative, r as resolveWorktreePath, s as isTransientTransportFailure, t as WorktreeAdapterError, u as openSearchLedger, v as ProfileMatrixError, w as userStoryScoreboard, x as renderScoreboardMarkdown, y as runProfileMatrix } from "../campaign-Dg2B4W6d.js";
7
+ import { A as neutralizationGate, C as scoreboardSummary, D as LabeledScenarioStoreError, E as FsLabeledScenarioStore, F as analyzeCrossSurfaceInteractions, I as buildTraceAnalystSurfaceDispatch, L as traceAnalystQualityJudge, M as loadEvalFixture, N as loadEvalFixtureScenarios, O as classifyUngroundedLiterals, P as planEvalFixtureRun, S as scoreUserStory, T as neutralizeText, _ as runProfileMatrixSegment, a as autoevalsScorerJudge, b as makePlaybackDispatch, c as FileSearchLedger, d as validateSearchLedgerEvent, f as scoreDiscrimination, g as finalizeProfileMatrix, h as createProfileMatrixPlan, i as verifyCodeSurface, j as discoverEvalFixtures, k as rolloutArgumentDiff, l as SEARCH_LEDGER_SCHEMA, m as SegmentedProfileMatrixError, n as gitWorktreeAdapter, o as phoenixEvaluatorJudge, p as selectDiscriminative, r as resolveWorktreePath, s as isTransientTransportFailure, t as WorktreeAdapterError, u as openSearchLedger, v as ProfileMatrixError, w as userStoryScoreboard, x as renderScoreboardMarkdown, y as runProfileMatrix } from "../campaign-B3GJpJ4h.js";
8
8
  import { n as paretoPolicy, r as paretoSignificanceGate, t as buildEvidenceVector } from "../promotion-policy-xzA40Evo.js";
9
9
  import { n as sequentialPairedGate, t as sequentialDecide } from "../sequential-C458DXNf.js";
10
10
  export { DEFAULT_EXTERNAL_OPTIMIZER_CALLBACK_LIMITS, DEFAULT_EXTERNAL_OPTIMIZER_PROCESS_LIMITS, FileSearchLedger, FsLabeledScenarioStore, LabeledScenarioStoreError, ProfileMatrixError, SEARCH_LEDGER_SCHEMA, SearchLedgerConflictError, SearchLedgerError, SearchLedgerIntegrityError, SegmentedProfileMatrixError, WorktreeAdapterError, acquireSingleRunLock, analyzeCrossSurfaceInteractions, assertCampaignDesign, assertCampaignSplitIdentity, assertCodeSurfaceIdentity, assertComponentSurface, autoevalsScorerJudge, buildEvidenceVector, buildLoopProvenanceRecord, buildTraceAnalystSurfaceDispatch, campaignBreakdown, campaignMeanComposite, campaignMeasurementDigest, campaignScenarioIdentity, campaignSplitDigest, campaignSplitDigestFromIdentities, canonicalDigest, cellCachePath, classifyUngroundedLiterals, codeSurfaceIdentityMaterial, combineComparisonCosts, compareOptimizationMethods, compareRankKeys, componentSurfaceIdentityMaterial, composeGate, costFromLedgerSummary, createProfileMatrixPlan, createReferenceEquivalenceJudge, createRunCostLedger, decodeExternalTextCandidate, defaultProductionGate, detectScale, dimensionRegressions, discoverEvalFixtures, emitLoopProvenance, externalTextOptimizationMethod, finalizeProfileMatrix, fsCampaignStorage, gepaOptimizationMethod, gitWorktreeAdapter, heldOutGate, heldoutSignificance, inMemoryCampaignStorage, isProposedCandidate, isTransientTransportFailure, labelTrustRank, llmJudge, loadEvalFixture, loadEvalFixtureScenarios, loopProvenanceArgsFromResult, loopProvenanceSpans, makePlaybackDispatch, makeProposalFinding, neutralizationGate, neutralizeText, openAutoPr, openSearchLedger, optimizationTokenUsageFromSummary, pairHoldout, paretoPolicy, paretoSignificanceGate, phoenixEvaluatorJudge, planCampaignRun, planEvalFixtureRun, powerPreflight, provenanceRecordPath, provenanceSpansPath, readCachedCell, readExternalOptimizerObservationArtifact, readGepaCandidatePopulationArtifact, renderScoreboardMarkdown, renderSurfaceDiff, resolveExternalOptimizerCallbackLimits, resolveExternalOptimizerProcessLimits, resolveRunDir, resolveWorktreePath, rolloutArgumentDiff, runCampaign, runEval, runImprovementLoop, runOptimization, runProfileMatrix, runProfileMatrixSegment, scoreDiscrimination, scoreUserStory, scoreboardSummary, selectDiscriminative, sequentialDecide, sequentialPairedGate, skillOptOptimizationMethod, surfaceContentHash, surfaceHash, tangleTracesRoot, traceAnalystQualityJudge, userStoryScoreboard, validateSearchLedgerEvent, verifyCodeSurface, verifyLoopProvenanceRecord };
@@ -1542,10 +1542,13 @@ function receiptModels(cell) {
1542
1542
  const reported = cell.resolvedModels ?? (cell.resolvedModel ? [cell.resolvedModel] : []);
1543
1543
  return [...new Set(reported.map((model) => model.trim()).filter(Boolean))];
1544
1544
  }
1545
+ function isUnknownModelOnly(reported) {
1546
+ return reported.length > 0 && reported.every((model) => model === "unknown");
1547
+ }
1545
1548
  /** Resolve and validate every paid-call model used by one cell. */
1546
1549
  function recordModel(cell, profileId, declaredModel) {
1547
1550
  const reported = receiptModels(cell);
1548
- if (reported.length === 0 && cell.error !== void 0) return UNKNOWN_MODEL;
1551
+ if (cell.error !== void 0 && (reported.length === 0 || isUnknownModelOnly(reported))) return UNKNOWN_MODEL;
1549
1552
  if (modelHasSnapshot(declaredModel)) {
1550
1553
  for (const model of reported) if (model !== declaredModel) throw new ProfileMatrixError(`profile '${profileId}' paid-call model '${model}' for cell '${cell.cellId}' does not match its declared exact model '${declaredModel}'`);
1551
1554
  return declaredModel;
@@ -1595,7 +1598,7 @@ function buildRunRecord(args) {
1595
1598
  */
1596
1599
  function durableCellForRecord(cell) {
1597
1600
  if (cell.error === void 0) return cell;
1598
- const hasSettledAgentCalls = receiptModels(cell).length > 0;
1601
+ const hasSettledAgentCalls = receiptModels(cell).some((model) => model !== UNKNOWN_MODEL);
1599
1602
  return {
1600
1603
  ...cell,
1601
1604
  tokenUsage: cell.tokenUsage.tokensKnown === false || hasSettledAgentCalls ? cell.tokenUsage : {
@@ -3682,4 +3685,4 @@ function resolveWorktreePath(surface, worktreeDir) {
3682
3685
  //#endregion
3683
3686
  export { neutralizationGate as A, scoreboardSummary as C, LabeledScenarioStoreError as D, FsLabeledScenarioStore as E, analyzeCrossSurfaceInteractions as F, buildTraceAnalystSurfaceDispatch as I, traceAnalystQualityJudge as L, loadEvalFixture as M, loadEvalFixtureScenarios as N, classifyUngroundedLiterals as O, planEvalFixtureRun as P, scoreUserStory as S, neutralizeText as T, runProfileMatrixSegment as _, autoevalsScorerJudge as a, makePlaybackDispatch as b, FileSearchLedger as c, validateSearchLedgerEvent as d, scoreDiscrimination as f, finalizeProfileMatrix as g, createProfileMatrixPlan as h, verifyCodeSurface as i, discoverEvalFixtures as j, rolloutArgumentDiff as k, SEARCH_LEDGER_SCHEMA as l, SegmentedProfileMatrixError as m, gitWorktreeAdapter as n, phoenixEvaluatorJudge as o, selectDiscriminative as p, resolveWorktreePath as r, isTransientTransportFailure as s, WorktreeAdapterError as t, openSearchLedger as u, ProfileMatrixError as v, userStoryScoreboard as w, renderScoreboardMarkdown as x, runProfileMatrix as y };
3684
3687
 
3685
- //# sourceMappingURL=campaign-Dg2B4W6d.js.map
3688
+ //# sourceMappingURL=campaign-B3GJpJ4h.js.map