@tangle-network/agent-eval 0.133.3 → 0.134.1

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (73) hide show
  1. package/CHANGELOG.md +70 -0
  2. package/dist/analyst/index.d.ts +11 -35
  3. package/dist/analyst/index.d.ts.map +1 -1
  4. package/dist/analyst/index.js +4 -53
  5. package/dist/analyst/index.js.map +1 -1
  6. package/dist/{analyze-runs-BClW9OSe.d.ts → analyze-runs-DMo3Lb_y.d.ts} +4 -4
  7. package/dist/{analyze-runs-BClW9OSe.d.ts.map → analyze-runs-DMo3Lb_y.d.ts.map} +1 -1
  8. package/dist/benchmarks/index.d.ts +1 -1
  9. package/dist/benchmarks/index.js +1 -1
  10. package/dist/{benchmarks-BP9sgMia.js → benchmarks-v5piCeDl.js} +3 -3
  11. package/dist/{benchmarks-BP9sgMia.js.map → benchmarks-v5piCeDl.js.map} +1 -1
  12. package/dist/campaign/index.d.ts +5 -4
  13. package/dist/campaign/index.js +4 -3
  14. package/dist/{campaign--V4ffEKR.js → campaign-DEC_7DLn.js} +2 -2
  15. package/dist/{campaign--V4ffEKR.js.map → campaign-DEC_7DLn.js.map} +1 -1
  16. package/dist/{client-Du7B81wW.d.ts → client-BIyh1RCr.d.ts} +3 -3
  17. package/dist/{client-Du7B81wW.d.ts.map → client-BIyh1RCr.d.ts.map} +1 -1
  18. package/dist/contract/index.d.ts +12 -11
  19. package/dist/contract/index.d.ts.map +1 -1
  20. package/dist/contract/index.js +4 -11
  21. package/dist/contract/index.js.map +1 -1
  22. package/dist/{default-registry-Cl3pHo4n.d.ts → default-registry-Brxr728w.d.ts} +4 -268
  23. package/dist/default-registry-Brxr728w.d.ts.map +1 -0
  24. package/dist/{default-registry-D3T9XbuY.js → default-registry-IjYs7T8l.js} +4 -61
  25. package/dist/default-registry-IjYs7T8l.js.map +1 -0
  26. package/dist/hosted/index.d.ts +2 -2
  27. package/dist/{index-DOqvIJ8I.d.ts → index-BoJNQR6n.d.ts} +4 -3
  28. package/dist/index-BoJNQR6n.d.ts.map +1 -0
  29. package/dist/{index-B5MNN1f1.d.ts → index-C21xKtxu.d.ts} +4 -4
  30. package/dist/{index-B5MNN1f1.d.ts.map → index-C21xKtxu.d.ts.map} +1 -1
  31. package/dist/index.d.ts +13 -12
  32. package/dist/index.d.ts.map +1 -1
  33. package/dist/index.js +190 -20
  34. package/dist/index.js.map +1 -1
  35. package/dist/multishot/index.d.ts +1 -1
  36. package/dist/openapi.json +1 -1
  37. package/dist/proposal-findings-DCawte-y.js +164 -0
  38. package/dist/proposal-findings-DCawte-y.js.map +1 -0
  39. package/dist/{release-report-DKBtegGt.d.ts → release-report-CuULWKyk.d.ts} +2 -2
  40. package/dist/{release-report-DKBtegGt.d.ts.map → release-report-CuULWKyk.d.ts.map} +1 -1
  41. package/dist/reporting.d.ts +2 -2
  42. package/dist/{researcher-BtD5U1Up.d.ts → researcher-DVtruQ9U.d.ts} +2 -2
  43. package/dist/{researcher-BtD5U1Up.d.ts.map → researcher-DVtruQ9U.d.ts.map} +1 -1
  44. package/dist/rl.d.ts +2 -2
  45. package/dist/rl.d.ts.map +1 -1
  46. package/dist/rl.js +14 -1
  47. package/dist/rl.js.map +1 -1
  48. package/dist/{semantic-concept-judge-BypLt6Fw.js → semantic-concept-judge-C0P1VTXD.js} +2 -3
  49. package/dist/{semantic-concept-judge-BypLt6Fw.js.map → semantic-concept-judge-C0P1VTXD.js.map} +1 -1
  50. package/dist/{skill-usage-BaaxFSJR.d.ts → skill-usage-BDQVPIG1.d.ts} +3 -2
  51. package/dist/skill-usage-BDQVPIG1.d.ts.map +1 -0
  52. package/dist/{skillopt-optimization-method-vvJ4bMNI.js → skillopt-optimization-method-BY6vKLJB.js} +51 -30
  53. package/dist/skillopt-optimization-method-BY6vKLJB.js.map +1 -0
  54. package/dist/{skillopt-optimization-method-Dxr8pdZd.d.ts → skillopt-optimization-method-DJ3l4w8W.d.ts} +10 -13
  55. package/dist/skillopt-optimization-method-DJ3l4w8W.d.ts.map +1 -0
  56. package/dist/{summary-report-DyOhItws.d.ts → summary-report-DGp0-_XO.d.ts} +59 -3
  57. package/dist/summary-report-DGp0-_XO.d.ts.map +1 -0
  58. package/dist/types-DVjczBM9.d.ts +276 -0
  59. package/dist/types-DVjczBM9.d.ts.map +1 -0
  60. package/dist/{types-BokuXvOG.d.ts → types-DiWLru6Z.d.ts} +20 -37
  61. package/dist/types-DiWLru6Z.d.ts.map +1 -0
  62. package/docs/campaign-proposers.md +5 -0
  63. package/package.json +1 -1
  64. package/dist/default-registry-Cl3pHo4n.d.ts.map +0 -1
  65. package/dist/default-registry-D3T9XbuY.js.map +0 -1
  66. package/dist/index-DOqvIJ8I.d.ts.map +0 -1
  67. package/dist/run-score-iEEAWiBY.js +0 -41
  68. package/dist/run-score-iEEAWiBY.js.map +0 -1
  69. package/dist/skill-usage-BaaxFSJR.d.ts.map +0 -1
  70. package/dist/skillopt-optimization-method-Dxr8pdZd.d.ts.map +0 -1
  71. package/dist/skillopt-optimization-method-vvJ4bMNI.js.map +0 -1
  72. package/dist/summary-report-DyOhItws.d.ts.map +0 -1
  73. package/dist/types-BokuXvOG.d.ts.map +0 -1
package/dist/index.js CHANGED
@@ -2,16 +2,16 @@ import { t as __exportAll } from "./rolldown-runtime-8H4AJuhK.js";
2
2
  import { _ as observeAll, a as jsonlReviewStore, c as scoreFromEvals, d as runAgentControlLoop, f as stopOnNoProgress, g as noProgressDetector, h as errorStreakDetector, i as inMemoryReviewStore, l as allCriticalPassed, m as subjectiveEval, n as runProposeReviewAsControlLoop, o as runProposeReview, p as stopOnRepeatedAction, r as createLlmReviewer, s as controlRunToRunRecord, t as controlFailureClassFromVerification, u as objectiveEval, v as repeatedActionDetector, y as evaluateActionPolicy } from "./propose-review-control-SQ-n9-We.js";
3
3
  import { a as NotFoundError, c as VerificationError, i as JudgeError, n as CaptureIntegrityError, o as ReplayError, r as ConfigError, s as ValidationError, t as AgentEvalError } from "./errors-8YnH8WlF.js";
4
4
  import { _ as verifyManifest, a as assertRunAgentProfileCell, c as groupRunsByAgentProfileCell, d as validateAgentProfileCell, f as verifyAgentProfileCell, g as signManifest, h as hashJson, i as agentProfileCellKey, l as requireAgentProfileCell, m as evaluateHypothesis, n as AgentProfileCellValidationError, o as buildAgentInterfaceProfileCell, p as canonicalize, r as agentProfileCellHashMaterial, s as buildAgentProfileCell, t as AGENT_PROFILE_KINDS, u as toAgentProfileJson } from "./agent-profile-cell-OhuTee9n.js";
5
- import { F as makeFinding, I as computeTraceMetrics, L as createChatClient, P as computeFindingId, R as createAnalystAi, a as KNOWLEDGE_GAP_KIND_SPEC, d as renderUpstreamFindings, i as KNOWLEDGE_POISONING_KIND_SPEC, l as createTraceAnalystKind, n as AnalystRegistry, o as IMPROVEMENT_KIND_SPEC, r as DEFAULT_TRACE_ANALYST_KINDS, s as FAILURE_MODE_KIND_SPEC, t as buildDefaultAnalystRegistry, u as renderPriorFindings } from "./default-registry-D3T9XbuY.js";
5
+ import { F as createChatClient, I as createAnalystAi, P as computeTraceMetrics, a as KNOWLEDGE_GAP_KIND_SPEC, d as renderUpstreamFindings, i as KNOWLEDGE_POISONING_KIND_SPEC, l as createTraceAnalystKind, n as AnalystRegistry, o as IMPROVEMENT_KIND_SPEC, r as DEFAULT_TRACE_ANALYST_KINDS, s as FAILURE_MODE_KIND_SPEC, t as buildDefaultAnalystRegistry, u as renderPriorFindings } from "./default-registry-IjYs7T8l.js";
6
6
  import { _ as resolveModelPricing, a as CostLedgerPersistenceError, c as costForTokenPricing, d as MODEL_PRICING, f as MetricsCollector, g as isModelPriced, h as estimateTokens, i as CostLedger, l as costForUsage, m as estimateCost, n as CostCallConflictError, o as CostReceiptCaptureError, p as TokenCounter, r as CostCeilingReachedError, s as CostReservationExceededError, t as CostAccountingIncompleteError, u as modelPriceKey } from "./cost-ledger-BrJxbrMy.js";
7
7
  import { a as providerFromBaseUrl, i as defaultProviderRedactor, n as InMemoryRawProviderSink, r as NoopRawProviderSink, t as FileSystemRawProviderSink } from "./raw-provider-sink-BQd7mzyT.js";
8
8
  import { a as assertLlmRoute, c as callLlmJson, d as isTransientLlmError, f as maximumChargeForLlmRequest, i as LlmRouteAssertionError, l as costReceiptFromLlm, m as stripFencedJson, n as LlmClient, o as backoffMs, p as probeLlm, r as LlmResponseError, s as callLlm, t as LlmCallError, u as costReceiptFromLlmError } from "./llm-client-ClPW-dWB.js";
9
9
  import { INPUT_VALUE, LLM_CACHED_TOKENS, LLM_CACHED_TOKEN_ATTR_KEYS, LLM_CACHE_WRITE_TOKENS, LLM_CACHE_WRITE_TOKEN_ATTR_KEYS, LLM_CONTEXT_TOKENS, LLM_COST_ATTR_KEYS, LLM_COST_USD, LLM_INPUT_TOKENS, LLM_INPUT_TOKEN_ATTR_KEYS, LLM_MODEL_ATTR_KEYS, LLM_MODEL_NAME, LLM_OUTPUT_TOKENS, LLM_OUTPUT_TOKEN_ATTR_KEYS, LLM_REASONING_TOKENS, LLM_REASONING_TOKEN_ATTR_KEYS, OPENINFERENCE_SPAN_KIND, OUTPUT_VALUE, RUN_COST_ATTR_KEYS, SPAN_KIND_ATTR_KEYS, TOOL_ARGS_CAPTURED, TOOL_LATENCY_MS, TOOL_NAME, TOOL_NAME_ATTR_KEYS, applyLlmSpanOtlpAttributes, asNumber, contextInputTokens, firstNumberAttr } from "./trace-attributes.js";
10
10
  import { S as traceSpanKindToOpenInferenceKind, a as SpanNotFoundError, b as classifyOtlpSpanRole, c as DEFAULT_TRACE_ANALYST_BUDGETS, f as extractOtlpAttributes, g as readOtlpStatus, h as projectOtlpFlatLine, i as OtlpFileTraceStore, l as TRACE_ANALYST_TRUNCATION_MARKER_PREFIX, m as inferOtlpKind, n as traceAnalystFunctionGroup, o as TraceFileMissingError, p as firstStringAttr, s as TraceNotFoundError, t as buildTraceAnalystTools, u as asString, v as stringField, x as isOtlpModelCall, y as applyToolSpanOtlpAttributes } from "./tools-BmuN627J.js";
11
- import { n as aggregateRunScore, r as clamp01, t as DEFAULT_RUN_SCORE_WEIGHTS } from "./run-score-iEEAWiBY.js";
11
+ import { a as clamp01, c as makeFinding, i as aggregateRunScore, l as makeProposalFinding, r as DEFAULT_RUN_SCORE_WEIGHTS, s as computeFindingId } from "./proposal-findings-DCawte-y.js";
12
12
  import { n as mapConcurrent, t as Mutex } from "./concurrency-MUjT7VjM.js";
13
- import { a as RunCritic, d as defaultIsMaterial, f as diffFindings, i as runSemanticConceptJudge, n as SEMANTIC_CONCEPT_JUDGE_VERSION, o as SKILL_USAGE_ANALYST, p as LockedJsonlAppender, r as createSemanticConceptJudge, s as SkillUsageAnalyst, t as DEFAULT_COMPLEXITY_WEIGHTS, u as FindingsStore } from "./semantic-concept-judge-BypLt6Fw.js";
14
- import { At as dominates, B as surfaceContentHash, Bt as summarizeAgentReceiptIntegrity, Ct as llmJudge, D as runCanaries, Dt as fileVerdictCache, Et as contentHash, Ft as minimumPairsForPairedDeltaTest, G as DEFAULT_MUTATION_PRIMITIVES, Ht as JudgeParseError, It as pairedDeltaTest, K as buildReflectionPrompt, Lt as BackendIntegrityError, Mt as paretoFrontierWithCrowding, Nt as scalarScore, Ot as inMemoryVerdictCache, Rt as assertRealAgentReceipts, St as hashScenarios, Tt as canonicalJson, Vt as summarizeBackendIntegrity, _t as redTeamReport, bt as Dataset, dt as REFERENCE_EQUIVALENCE_INPUT_LIMITS, ft as REFERENCE_EQUIVALENCE_JUDGE_VERSION, gt as redTeamDataset, ht as DEFAULT_RED_TEAM_CORPUS, jt as paretoFrontier, kt as crowdingDistance, mt as runReferenceEquivalenceJudge, pt as createReferenceEquivalenceJudge, q as parseReflectionResponse, vt as scoreRedTeamOutput, wt as cachedJudge, xt as HoldoutLockedError, yt as toolNamesForRun, zt as assertRealBackend } from "./skillopt-optimization-method-vvJ4bMNI.js";
13
+ import { a as RunCritic, d as defaultIsMaterial, f as diffFindings, i as runSemanticConceptJudge, n as SEMANTIC_CONCEPT_JUDGE_VERSION, o as SKILL_USAGE_ANALYST, p as LockedJsonlAppender, r as createSemanticConceptJudge, s as SkillUsageAnalyst, t as DEFAULT_COMPLEXITY_WEIGHTS, u as FindingsStore } from "./semantic-concept-judge-C0P1VTXD.js";
14
+ import { At as dominates, B as surfaceContentHash, Bt as summarizeAgentReceiptIntegrity, Ct as llmJudge, D as runCanaries, Dt as fileVerdictCache, Et as contentHash, Ft as minimumPairsForPairedDeltaTest, G as DEFAULT_MUTATION_PRIMITIVES, Ht as JudgeParseError, It as pairedDeltaTest, K as buildReflectionPrompt, Lt as BackendIntegrityError, Mt as paretoFrontierWithCrowding, Nt as scalarScore, Ot as inMemoryVerdictCache, Rt as assertRealAgentReceipts, St as hashScenarios, Tt as canonicalJson, Vt as summarizeBackendIntegrity, _t as redTeamReport, bt as Dataset, dt as REFERENCE_EQUIVALENCE_INPUT_LIMITS, ft as REFERENCE_EQUIVALENCE_JUDGE_VERSION, gt as redTeamDataset, ht as DEFAULT_RED_TEAM_CORPUS, jt as paretoFrontier, kt as crowdingDistance, mt as runReferenceEquivalenceJudge, pt as createReferenceEquivalenceJudge, q as parseReflectionResponse, vt as scoreRedTeamOutput, wt as cachedJudge, xt as HoldoutLockedError, yt as toolNamesForRun, zt as assertRealBackend } from "./skillopt-optimization-method-BY6vKLJB.js";
15
15
  import { A as passAtK, B as studentTCdf, C as pairedBootstrap, D as pairedSignTest, E as pairedRiskDifference, F as spearmanR, G as continuousAgreement, H as normalCdf, I as weightedComposite, J as verbosityBias, K as positionalBias, L as weightedMean, M as ranks, N as requiredPairedSampleSize, O as pairedTTest, P as requiredSampleSize, R as wilcoxonSignedRank, S as normalizeScores, T as pairedMde, U as calibrateJudge, W as calibrateJudgeContinuous, _ as mannWhitneyU, a as WILCOXON_EXACT_MAX_N, b as mcnemarRequiredN, c as cliffsDelta, d as corpusInterRaterAgreement, f as corpusInterRaterAgreementFromJudgeScores, g as interpretCliffs, h as interRaterReliability, i as MANN_WHITNEY_EXACT_MAX_WORK, j as pearsonR, k as partialCredit, l as cohensD, m as holm, n as DEFAULT_PERMUTATIONS, o as benjaminiHochberg, p as eProcess, q as selfPreference, r as MANN_WHITNEY_EXACT_MAX_STATES, s as bonferroni, t as BOOTSTRAP_GATE_MIN_N, u as confidenceInterval, v as mcnemar, w as pairedCohensDz, x as mulberry32, y as mcnemarPower, z as wilson } from "./statistics-RwRNu2__.js";
16
16
  import { n as pairArms, r as pairRunRecords, t as comparePairedArms } from "./paired-arms-CA_8pN01.js";
17
17
  import { n as llmSpanFromProvider, t as TraceEmitter } from "./emitter-CPBAhxum.js";
@@ -28,7 +28,7 @@ import { D as showMeasured, E as isUnavailable, S as rollupSupervisorRuns, T as
28
28
  import { n as TRACE_ANALYST_ACTOR_DESCRIPTION, r as TRACE_ANALYST_ACTOR_DESCRIPTION_VERSION, t as analyzeTraces } from "./analyst-LsnNpSkm.js";
29
29
  import { C as planTraceInsightQuestions, E as traceAnalystOnRunComplete, S as inferDomainKeywords, T as tokenizeDomainWords, _ as buildTraceInsightContext, a as convertTraceStoresToOtlp, b as describeTraceInsightScope, c as otelRunCompleteHook, d as captureFetchToRawSink, f as otlpRowsToRunRecords, g as flattenOtlpExportToNdjson, h as otlpToTraceRunRecords, i as iterateRawCalls, l as OTEL_AGENT_EVAL_SCOPE, m as otlpToRunRecords, n as ReplayCacheMissError, o as createOtelExporter, p as otlpRowsToTraceRunRecords, r as createReplayFetch, s as createOtelTracingStore, t as ReplayCache, u as exportRunAsOtlp, v as buildTraceInsightPrompt, w as scoreTraceInsightReadiness, x as domainEvidencePattern, y as defaultTraceInsightPanel } from "./replay-CJfGLdx4.js";
30
30
  import { n as extractUsageFromResponse, r as extractUsageFromSse, t as extractUsage } from "./extract-usage-2j25whHw.js";
31
- import { B as harnessAxisOf, F as HARNESS_NATIVE_MODEL, G as parseCorrectnessResponse, H as completionVerdict, I as agentProfileHash, K as verifyCompletion, L as agentProfileId, P as CODING_HARNESSES, R as agentProfileModelId, U as createLlmCorrectnessChecker, V as extractProducedState, W as createTokenRecallChecker, z as expandProfileAxes } from "./campaign--V4ffEKR.js";
31
+ import { B as harnessAxisOf, F as HARNESS_NATIVE_MODEL, G as parseCorrectnessResponse, H as completionVerdict, I as agentProfileHash, K as verifyCompletion, L as agentProfileId, P as CODING_HARNESSES, R as agentProfileModelId, U as createLlmCorrectnessChecker, V as extractProducedState, W as createTokenRecallChecker, z as expandProfileAxes } from "./campaign-DEC_7DLn.js";
32
32
  import { n as iqr, r as welchsTTest, t as compareToBaseline } from "./baseline-BaPxoROc.js";
33
33
  import { a as judgeSpans, c as runsForScenario, i as hasCapturedToolArgs, l as toolSpans, n as argHash, o as llmSpans, r as groupBy, s as runFailureClass, t as aggregateLlm } from "./query-Di7eEQ79.js";
34
34
  import { a as checkBehavioralCanary, i as canaryLeakView, o as checkCanaries, r as HoldoutAuditor, s as runBehavioralCanaries, t as analyzeRuns } from "./analyze-runs-qk8op0tN.js";
@@ -38,7 +38,7 @@ import { a as InMemoryTraceStore, i as FileSystemTraceStore, n as assertRunCaptu
38
38
  import { i as redactValue, n as REDACTION_VERSION, r as redactString, t as DEFAULT_REDACTION_RULES } from "./redact-7Aq1ukl-.js";
39
39
  import { n as DEFAULT_RULES, r as classifyFailure, t as computeToolUseMetrics } from "./tool-use-metrics-DEGMKycK.js";
40
40
  import { t as analyzeSeries } from "./series-convergence-CjO2QdRW.js";
41
- import { _ as deterministicSplit, g as BENCHMARK_SPLIT_SEED, t as benchmarks_exports } from "./benchmarks-BP9sgMia.js";
41
+ import { _ as deterministicSplit, g as BENCHMARK_SPLIT_SEED, t as benchmarks_exports } from "./benchmarks-v5piCeDl.js";
42
42
  import { t as runEvalCampaign } from "./eval-campaign-CvPcvqXC.js";
43
43
  import { n as pairedEvalueSequence, t as evaluateInterimReleaseConfidence } from "./sequential-Br0mAPHA.js";
44
44
  import { AxGEPA, AxSignature, ax } from "@ax-llm/ax";
@@ -6792,11 +6792,19 @@ async function evaluateContract(store, contract) {
6792
6792
  pass: false
6793
6793
  };
6794
6794
  const samples = [];
6795
+ const measurementBreaches = [];
6795
6796
  for (const m of contract.metrics) {
6796
6797
  const extract = m.extract ?? defaultExtract$1(m.metric);
6797
6798
  const baseline = await extractAll(baselineRuns, extract, store);
6798
6799
  const candidate = await extractAll(candidateRuns, extract, store);
6799
- if (baseline.length < 2 || candidate.length < 2) continue;
6800
+ if (baseline.length !== baselineRuns.length || candidate.length !== candidateRuns.length) {
6801
+ measurementBreaches.push(`metric "${m.metric}" measured ${candidate.length}/${candidateRuns.length} candidate and ${baseline.length}/${baselineRuns.length} baseline run(s); every supplied run must report the metric`);
6802
+ continue;
6803
+ }
6804
+ if (baseline.length < 2 || candidate.length < 2) {
6805
+ measurementBreaches.push(`metric "${m.metric}" has too few comparable samples: baseline ${baseline.length}, candidate ${candidate.length}; need at least 2 per arm`);
6806
+ continue;
6807
+ }
6800
6808
  samples.push({
6801
6809
  metric: m.metric,
6802
6810
  higherIsBetter: m.higherIsBetter,
@@ -6811,7 +6819,7 @@ async function evaluateContract(store, contract) {
6811
6819
  };
6812
6820
  let sloReport;
6813
6821
  if (contract.slos && contract.slos.length > 0) sloReport = checkSlos(await aggregateRunMetrics(candidateRuns, store), contract.slos);
6814
- const breaches = [];
6822
+ const breaches = [...measurementBreaches];
6815
6823
  for (const metric of baselineReport.metrics) {
6816
6824
  const decl = contract.metrics.find((m) => m.metric === metric.metric);
6817
6825
  if (!decl) continue;
@@ -9338,6 +9346,7 @@ function decideReferenceReplayPromotion(baseline, candidate, policy = {}) {
9338
9346
  const compared = comparisons.filter((item) => requiredSplits.includes(item.split));
9339
9347
  const regressions = comparisons.filter((item) => item.f1Delta < -maxRegression);
9340
9348
  const aggregateDelta = candidate.aggregate.f1 - baseline.aggregate.f1;
9349
+ const coverageMismatch = compareScenarioCoverage(baseline, candidate);
9341
9350
  if (missingRequiredSplits.length > 0) return {
9342
9351
  promote: false,
9343
9352
  reason: `Required split missing from baseline or candidate: ${missingRequiredSplits.join(", ")}`,
@@ -9366,6 +9375,16 @@ function decideReferenceReplayPromotion(baseline, candidate, policy = {}) {
9366
9375
  comparisons,
9367
9376
  regressions
9368
9377
  };
9378
+ if (coverageMismatch.candidateMissing.length > 0 || coverageMismatch.baselineMissing.length > 0) {
9379
+ const gaps = [coverageMismatch.candidateMissing.length > 0 ? `candidate missing ${formatScenarioIds(coverageMismatch.candidateMissing)}` : null, coverageMismatch.baselineMissing.length > 0 ? `baseline missing ${formatScenarioIds(coverageMismatch.baselineMissing)}` : null].filter((part) => part !== null);
9380
+ return {
9381
+ promote: false,
9382
+ reason: `Scenario coverage differs: baseline ${baseline.scenarios.length}, candidate ${candidate.scenarios.length}; ${gaps.join("; ")}`,
9383
+ aggregateDelta,
9384
+ comparisons,
9385
+ regressions
9386
+ };
9387
+ }
9369
9388
  const requiredMeanDelta = mean(compared.map((item) => item.f1Delta));
9370
9389
  if (requiredMeanDelta < minF1Delta) return {
9371
9390
  promote: false,
@@ -9382,6 +9401,35 @@ function decideReferenceReplayPromotion(baseline, candidate, policy = {}) {
9382
9401
  regressions
9383
9402
  };
9384
9403
  }
9404
+ function compareScenarioCoverage(baseline, candidate) {
9405
+ const baselineCounts = scenarioIdentityCounts(baseline);
9406
+ const candidateCounts = scenarioIdentityCounts(candidate);
9407
+ return {
9408
+ candidateMissing: countDifference(baselineCounts, candidateCounts),
9409
+ baselineMissing: countDifference(candidateCounts, baselineCounts)
9410
+ };
9411
+ }
9412
+ function scenarioIdentityCounts(score) {
9413
+ const counts = /* @__PURE__ */ new Map();
9414
+ for (const scenario of score.scenarios) {
9415
+ const identity = `${scenario.split}:${scenario.scenarioId}`;
9416
+ counts.set(identity, (counts.get(identity) ?? 0) + 1);
9417
+ }
9418
+ return counts;
9419
+ }
9420
+ function countDifference(expected, actual) {
9421
+ const missing = [];
9422
+ for (const [identity, count] of expected) {
9423
+ const difference = count - (actual.get(identity) ?? 0);
9424
+ for (let index = 0; index < difference; index++) missing.push(identity);
9425
+ }
9426
+ return missing.sort();
9427
+ }
9428
+ function formatScenarioIds(ids) {
9429
+ const shown = ids.slice(0, 5);
9430
+ const remaining = ids.length - shown.length;
9431
+ return remaining > 0 ? `${shown.join(", ")}, +${remaining} more` : shown.join(", ");
9432
+ }
9385
9433
  function defaultReferenceReplayMatcher(reference, candidate) {
9386
9434
  const textScore = tokenJaccard(`${reference.title} ${reference.description ?? ""}`, `${candidate.title} ${candidate.description ?? ""}`);
9387
9435
  const severityScore = reference.severity && candidate.severity ? normalize(reference.severity) === normalize(candidate.severity) ? .1 : -.05 : 0;
@@ -9837,13 +9885,15 @@ function runScore$1(run) {
9837
9885
  function taskKey(run) {
9838
9886
  return run.scenarioId;
9839
9887
  }
9840
- /** Mean per-task score for a model: { taskKey -> mean(score over its runs) }. */
9888
+ /** Tasks one arm was dealt and the mean score for each task it answered. */
9841
9889
  function perTaskMeanScore(runs) {
9890
+ const dealt = /* @__PURE__ */ new Set();
9842
9891
  const acc = /* @__PURE__ */ new Map();
9843
9892
  for (const run of runs) {
9893
+ const key = taskKey(run);
9894
+ dealt.add(key);
9844
9895
  const s = runScore$1(run);
9845
9896
  if (s === void 0) continue;
9846
- const key = taskKey(run);
9847
9897
  const cur = acc.get(key) ?? {
9848
9898
  sum: 0,
9849
9899
  n: 0
@@ -9852,7 +9902,10 @@ function perTaskMeanScore(runs) {
9852
9902
  cur.n += 1;
9853
9903
  acc.set(key, cur);
9854
9904
  }
9855
- return new Map([...acc].map(([k, v]) => [k, v.sum / v.n]));
9905
+ return {
9906
+ dealt,
9907
+ answered: new Map([...acc].map(([key, value]) => [key, value.sum / value.n]))
9908
+ };
9856
9909
  }
9857
9910
  /** Compressed-model bits — the model's description length L_model. gzip is a
9858
9911
  * deterministic, dependency-free stand-in for Kolmogorov complexity; it
@@ -9866,7 +9919,7 @@ function dataDescriptionBits(scoreByTask, keys, scoreFloor) {
9866
9919
  let bits = 0;
9867
9920
  for (const key of keys) {
9868
9921
  const s = scoreByTask.get(key);
9869
- if (s === void 0) continue;
9922
+ if (s === void 0) throw new ValidationError(`dataDescriptionBits: no score for task '${key}'`);
9870
9923
  bits += -Math.log2(Math.max(s, scoreFloor));
9871
9924
  }
9872
9925
  return bits;
@@ -9892,9 +9945,14 @@ var DescriptionLengthGate = class {
9892
9945
  * if λ·L_model + L_data is strictly lower by at least `marginBits`. */
9893
9946
  evaluate(candidate, baseline) {
9894
9947
  const candidateId = inferCandidateId$1(candidate.runs, this.baselineKey);
9895
- const candScores = perTaskMeanScore(candidate.runs);
9896
- const baseScores = perTaskMeanScore(baseline.runs);
9897
- const shared = [...candScores.keys()].filter((k) => baseScores.has(k));
9948
+ const candidateTasks = perTaskMeanScore(candidate.runs);
9949
+ const baselineTasks = perTaskMeanScore(baseline.runs);
9950
+ const candScores = candidateTasks.answered;
9951
+ const baseScores = baselineTasks.answered;
9952
+ const dealt = [.../* @__PURE__ */ new Set([...candidateTasks.dealt, ...baselineTasks.dealt])].sort();
9953
+ const shared = dealt.filter((key) => candScores.has(key) && baseScores.has(key));
9954
+ const candidateMissing = dealt.filter((key) => !candScores.has(key));
9955
+ const baselineMissing = dealt.filter((key) => !baseScores.has(key));
9898
9956
  const modelBits = {
9899
9957
  candidate: this.lambda * modelDescriptionBits(candidate.content),
9900
9958
  baseline: this.lambda * modelDescriptionBits(baseline.content)
@@ -9928,6 +9986,15 @@ var DescriptionLengthGate = class {
9928
9986
  reason: `few_tasks: ${shared.length} shared task(s) < min ${this.minTasks}`,
9929
9987
  rejectionCode: "few_tasks"
9930
9988
  };
9989
+ if (candidateMissing.length > 0 || baselineMissing.length > 0) {
9990
+ const missing = [candidateMissing.length > 0 ? `candidate answered ${candScores.size}/${dealt.length}, ${candidateMissing.length} missing: ${formatTaskIds(candidateMissing)}` : null, baselineMissing.length > 0 ? `baseline answered ${baseScores.size}/${dealt.length}, ${baselineMissing.length} missing: ${formatTaskIds(baselineMissing)}` : null].filter((part) => part !== null);
9991
+ return {
9992
+ ...base,
9993
+ promote: false,
9994
+ reason: `missing_task_scores: ${missing.join("; ")}. The decision cannot use only the ${shared.length} task(s) both arms answered`,
9995
+ rejectionCode: "missing_task_scores"
9996
+ };
9997
+ }
9931
9998
  if (!(evidence.deltaBits < -this.marginBits)) {
9932
9999
  const code = evidence.dataBitsDelta < 0 ? "model_bloat" : "no_total_gain";
9933
10000
  const why = code === "model_bloat" ? `model grew ${fmt(evidence.modelBitsDelta)} bits, outpacing a ${fmt(-evidence.dataBitsDelta)}-bit data gain` : `outcomes did not improve (data Δ=${fmt(evidence.dataBitsDelta)} bits)`;
@@ -9946,6 +10013,11 @@ var DescriptionLengthGate = class {
9946
10013
  };
9947
10014
  }
9948
10015
  };
10016
+ function formatTaskIds(ids) {
10017
+ const shown = ids.slice(0, 5);
10018
+ const remaining = ids.length - shown.length;
10019
+ return remaining > 0 ? `${shown.join(", ")}, +${remaining} more` : shown.join(", ");
10020
+ }
9949
10021
  function inferCandidateId$1(runs, baselineKey) {
9950
10022
  for (const run of runs) if (run.candidateId && run.candidateId !== baselineKey) return run.candidateId;
9951
10023
  return runs[0]?.candidateId ?? "(unknown candidate)";
@@ -10089,8 +10161,11 @@ function precision(goldens, candidates, options = {}) {
10089
10161
  * The optimizer's best-guess is one thing; what we should actually
10090
10162
  * ship is another. The gate is the line between them.
10091
10163
  *
10092
- * A candidate is promoted iff ALL three pass:
10164
+ * A candidate is promoted iff ALL FOUR pass:
10093
10165
  *
10166
+ * 0. **Coverage**: on BOTH splits, the fraction of DEALT work items that
10167
+ * carry a real score on BOTH arms is at least `minCoverage`
10168
+ * (default 1 — all of them). See "Missing scores are missing data".
10094
10169
  * 1. **Productive runs**: the candidate has at least
10095
10170
  * `minProductiveRuns` paired observations on items where BOTH
10096
10171
  * candidate and baseline produced a real (non-silent) score.
@@ -10103,6 +10178,36 @@ function precision(goldens, candidates, options = {}) {
10103
10178
  * "Better on search, worse on holdout" is the canonical
10104
10179
  * overfit pattern; this catches it.
10105
10180
  *
10181
+ * ## Missing scores are missing data
10182
+ *
10183
+ * Gates 1-3 all read numbers off the items where BOTH arms produced a finite
10184
+ * score. Without gate 0 that set is a SILENT FILTER: an item the candidate
10185
+ * crashed on, timed out on, or never wrote a row for simply left the
10186
+ * comparison, and the verdict was computed over whatever survived. Measured on
10187
+ * the pre-0.134 gate: 26 held-out items, a candidate that produced no score at
10188
+ * all on 20 of them, promoted at every threshold from −0.05 to +0.30 on the 6
10189
+ * it answered (`unpairedBaselineRuns: 20` sat in the evidence, read by
10190
+ * nothing); the control — the same 20 failures scored as the 0 they earned —
10191
+ * was correctly refused at a mean paired delta of −0.3808. An agent that fails
10192
+ * 77% of its tasks was promoted, and every guarantee gates 1-3 provide sat
10193
+ * behind that filter.
10194
+ *
10195
+ * The gate does NOT impute a value for those items. It cannot: it does not know
10196
+ * the failure value of the caller's metric (0 on [0,1]? 0 on 0-100? a
10197
+ * lower-is-better latency?), and guessing one is exactly the silent assumption
10198
+ * this gate exists to remove. A caller who DOES know it writes it onto the
10199
+ * record before calling — that is the caller's decision to make and to be
10200
+ * accountable for, and it then travels in the run corpus where an auditor can
10201
+ * see it. What the gate does instead is REFUSE, and report the full coverage
10202
+ * accounting (`GateEvidence.holdoutCoverage` / `searchCoverage`) so the caller
10203
+ * can see exactly which items went dark and on which arm.
10204
+ *
10205
+ * The denominator is MEASURED, never assumed: it is the set of work items the
10206
+ * comparison was DEALT, which is what `pairRunRecords` reports when it is given
10207
+ * every row of a split rather than only the scored ones. There is no list of
10208
+ * "expected" scenarios to keep in sync and nothing to trust — an item counts as
10209
+ * dealt because a row for it exists on at least one arm.
10210
+ *
10106
10211
  * The decision carries a machine-readable `rejectionCode` plus an
10107
10212
  * `evidence` block with every number the gate looked at, so the
10108
10213
  * downstream researcher / paper / dashboard can re-derive the
@@ -10128,8 +10233,11 @@ var HeldOutGate = class {
10128
10233
  resamples;
10129
10234
  seed;
10130
10235
  costPerTaskCeiling;
10236
+ minCoverage;
10131
10237
  constructor(config) {
10132
10238
  if (!config.baselineKey) throw new Error("HeldOutGate: baselineKey is required");
10239
+ if (config.minCoverage !== void 0 && !(Number.isFinite(config.minCoverage) && config.minCoverage >= 0 && config.minCoverage <= 1)) throw new Error(`HeldOutGate: minCoverage must be a finite fraction in [0, 1], got ${config.minCoverage}`);
10240
+ this.minCoverage = config.minCoverage ?? 1;
10133
10241
  this.confidence = config.confidence ?? .95;
10134
10242
  this.minProductiveRuns = Math.max(config.minProductiveRuns ?? 3, minimumPairsForPairedDeltaTest(this.confidence));
10135
10243
  this.pairedDeltaThreshold = config.pairedDeltaThreshold ?? 0;
@@ -10156,6 +10264,8 @@ var HeldOutGate = class {
10156
10264
  const baselineHoldout = scoredRuns(honestBaseline, "holdout");
10157
10265
  const searchPairing = pairRunRecords(baselineSearch, candidateSearch);
10158
10266
  const holdoutPairing = pairRunRecords(baselineHoldout, candidateHoldout);
10267
+ const searchCoverage = splitCoverage(honestBaseline, honestCandidate, "search");
10268
+ const holdoutCoverage = splitCoverage(honestBaseline, honestCandidate, "holdout");
10159
10269
  const splitScoreOf = (run, split) => observedSplitScore(run, split);
10160
10270
  const beforeSearch = searchPairing.pairs.map((pair) => splitScoreOf(pair.baseline, "search"));
10161
10271
  const afterSearch = searchPairing.pairs.map((pair) => splitScoreOf(pair.treatment, "search"));
@@ -10168,8 +10278,8 @@ var HeldOutGate = class {
10168
10278
  const baselineHoldoutMean = meanOrNull(beforeHoldout);
10169
10279
  const overfitGap = diffOrNull(candidateSearchMean, candidateHoldoutMean);
10170
10280
  const baselineOverfitGap = diffOrNull(baselineSearchMean, baselineHoldoutMean);
10171
- const medianCandidateCost = completeCostMedian(candidate);
10172
- const medianBaselineCost = completeCostMedian(baseline);
10281
+ const medianCandidateCost = completeCostMedian([...searchPairing.pairs.map((pair) => pair.treatment), ...holdoutPairing.pairs.map((pair) => pair.treatment)]);
10282
+ const medianBaselineCost = completeCostMedian([...searchPairing.pairs.map((pair) => pair.baseline), ...holdoutPairing.pairs.map((pair) => pair.baseline)]);
10173
10283
  const commonEvidence = {
10174
10284
  productiveRuns,
10175
10285
  unpairedCandidateRuns: holdoutPairing.unpairedTreatment.length,
@@ -10180,7 +10290,9 @@ var HeldOutGate = class {
10180
10290
  baselineOverfitGap,
10181
10291
  medianCandidateCost,
10182
10292
  medianBaselineCost,
10183
- realnessGatedRuns
10293
+ realnessGatedRuns,
10294
+ searchCoverage,
10295
+ holdoutCoverage
10184
10296
  };
10185
10297
  const missingSplitScores = [
10186
10298
  candidateSearch.length === 0 ? "candidate search" : null,
@@ -10201,6 +10313,20 @@ var HeldOutGate = class {
10201
10313
  reason: `missing_split_scores: ${missingSplitScores.join(", ")} score evidence is absent`,
10202
10314
  rejectionCode: "missing_split_scores"
10203
10315
  };
10316
+ const shortfalls = [["holdout", holdoutCoverage], ["search", searchCoverage]].filter(([, cov]) => cov.coverage < this.minCoverage);
10317
+ if (shortfalls.length > 0) return {
10318
+ promote: false,
10319
+ candidateId,
10320
+ baselineId,
10321
+ evidence: {
10322
+ ...commonEvidence,
10323
+ medianPairedDelta: productiveRuns > 0 ? medianDelta(beforeHoldout, afterHoldout) : null,
10324
+ pairedCI: null,
10325
+ pairedPValue: null
10326
+ },
10327
+ reason: `incomplete_coverage: ${shortfalls.map(([split, cov]) => describeCoverage(split, cov)).join("; ")} — required ${fmt(this.minCoverage)} of every dealt item scored on both arms. A missing score is missing data, not an absent item: the gate will not decide on the subset that answered`,
10328
+ rejectionCode: "incomplete_coverage"
10329
+ };
10204
10330
  if (productiveRuns < this.minProductiveRuns) return {
10205
10331
  promote: false,
10206
10332
  candidateId,
@@ -10304,6 +10430,50 @@ function scoredRuns(runs, split) {
10304
10430
  return typeof v === "number" && Number.isFinite(v);
10305
10431
  });
10306
10432
  }
10433
+ /** Every row tagged to one split, scored or not — the work the split was
10434
+ * DEALT, as opposed to the work it managed to measure. */
10435
+ function dealtRuns(runs, split) {
10436
+ return runs.filter((run) => run.splitTag === split);
10437
+ }
10438
+ function hasFiniteScore(run, split) {
10439
+ const v = observedSplitScore(run, split);
10440
+ return typeof v === "number" && Number.isFinite(v);
10441
+ }
10442
+ /**
10443
+ * Partition one split's dealt work into answered / unscored / one-armed.
10444
+ *
10445
+ * Pairs the UNFILTERED rows with the same `pairRunRecords` the decision uses,
10446
+ * so the denominator comes out of the same identity rules as the numerator and
10447
+ * cannot drift from it. Deliberately a SECOND pairing rather than a widening of
10448
+ * the decision's own: the decision keeps pairing exactly the rows it always
10449
+ * paired, so this check can only ever add a refusal, never move a verdict that
10450
+ * was already being made on a complete set.
10451
+ */
10452
+ function splitCoverage(baseline, candidate, split) {
10453
+ const pairing = pairRunRecords(dealtRuns(baseline, split), dealtRuns(candidate, split));
10454
+ let answered = 0;
10455
+ for (const pair of pairing.pairs) if (hasFiniteScore(pair.baseline, split) && hasFiniteScore(pair.treatment, split)) answered += 1;
10456
+ const unscoredPairs = pairing.pairs.length - answered;
10457
+ const candidateOnly = pairing.unpairedTreatment.length;
10458
+ const baselineOnly = pairing.unpairedBaseline.length;
10459
+ const dealt = pairing.pairs.length + candidateOnly + baselineOnly;
10460
+ return {
10461
+ dealt,
10462
+ answered,
10463
+ unscoredPairs,
10464
+ candidateOnly,
10465
+ baselineOnly,
10466
+ coverage: dealt === 0 ? 0 : answered / dealt
10467
+ };
10468
+ }
10469
+ function describeCoverage(split, cov) {
10470
+ const dark = [
10471
+ cov.unscoredPairs > 0 ? `${cov.unscoredPairs} ran on both arms but produced no score` : null,
10472
+ cov.baselineOnly > 0 ? `${cov.baselineOnly} only the baseline produced a row for` : null,
10473
+ cov.candidateOnly > 0 ? `${cov.candidateOnly} only the candidate produced a row for` : null
10474
+ ].filter((part) => part !== null);
10475
+ return `${split} scored ${cov.answered} of ${cov.dealt} dealt items (coverage ${fmt(cov.coverage)})` + (dark.length > 0 ? ` — ${dark.join(", ")}` : "");
10476
+ }
10307
10477
  function meanOrNull(xs) {
10308
10478
  if (xs.length === 0) return null;
10309
10479
  return xs.reduce((s, x) => s + x, 0) / xs.length;
@@ -12123,6 +12293,6 @@ function assertProductBenchmarkRun(runDir) {
12123
12293
  return report;
12124
12294
  }
12125
12295
  //#endregion
12126
- export { AGENT_PROFILE_KINDS, ATIF_SCHEMA_VERSION, ATTESTATION_ALGORITHM, AgentDriver, AgentEvalError, AgentProfileCellValidationError, AnalystRegistry, AxGepaSteeringOptimizer, BENCHMARK_SPLIT_SEED, BOOTSTRAP_GATE_MIN_N, BackendIntegrityError, BenchmarkRunner, BudgetBreachError, BudgetGuard, CODING_HARNESSES, CallbackResearcher, CaptureIntegrityError, ConfigError, ConvergenceTracker, CostAccountingIncompleteError, CostCallConflictError, CostCeilingReachedError, CostLedger, CostLedgerPersistenceError, CostReceiptCaptureError, CostReservationExceededError, CostTracker, CrossFamilyError, DEFAULT_AGENT_SLOS, DEFAULT_COMPLEXITY_WEIGHTS, DEFAULT_RULES as DEFAULT_FAILURE_RULES, DEFAULT_FINDERS, DEFAULT_HARNESS_OBJECTIVES, DEFAULT_MUTATION_PRIMITIVES, DEFAULT_MUTATORS, DEFAULT_PERMUTATIONS, DEFAULT_PR_REVIEW_SCORE_WEIGHTS, DEFAULT_REDACTION_RULES, DEFAULT_RED_TEAM_CORPUS, DEFAULT_RUN_SCORE_WEIGHTS, DEFAULT_SEVERITY_WEIGHTS, DEFAULT_TRACE_ANALYST_BUDGETS, DEFAULT_TRACE_ANALYST_KINDS, Dataset, DescriptionLengthGate, DockerSandboxDriver, DualAgentBench, ERROR_COUNT_PATTERNS, EvalTraceStore, ExperimentTracker, FAILURE_CLASSES, FAILURE_MODE_KIND_SPEC, FileSystemFeedbackTrajectoryStore, FileSystemRawProviderSink, FileSystemTraceStore, FindingsStore, HARBOR_IMPORT_GAP, HARNESS_NATIVE_MODEL, HeldOutGate, HoldoutAuditor, HoldoutLockedError, IMPROVEMENT_KIND_SPEC, INPUT_VALUE, INTENT_MATCH_JUDGE_VERSION, InMemoryFeedbackTrajectoryStore, InMemoryRawProviderSink, InMemoryTraceStore, InMemoryWorkspaceInspector, JudgeError, JudgeParseError, JudgeRunner, KNOWLEDGE_GAP_KIND_SPEC, KNOWLEDGE_POISONING_KIND_SPEC, LLM_CACHED_TOKENS, LLM_CACHED_TOKEN_ATTR_KEYS, LLM_CACHE_WRITE_TOKENS, LLM_CACHE_WRITE_TOKEN_ATTR_KEYS, LLM_CONTEXT_TOKENS, LLM_COST_ATTR_KEYS, LLM_COST_USD, LLM_INPUT_TOKENS, LLM_INPUT_TOKEN_ATTR_KEYS, LLM_MODEL_ATTR_KEYS, LLM_MODEL_NAME, LLM_OUTPUT_TOKENS, LLM_OUTPUT_TOKEN_ATTR_KEYS, LLM_REASONING_TOKENS, LLM_REASONING_TOKEN_ATTR_KEYS, LlmCallError, LlmClient, LlmResponseError, LlmRouteAssertionError, LockedJsonlAppender, MANN_WHITNEY_EXACT_MAX_STATES, MANN_WHITNEY_EXACT_MAX_WORK, MODEL_PRICING, MetricsCollector, ModelsUnreachableError, MultiLayerVerifier, Mutex, NoopRawProviderSink, NoopResearcher, NotFoundError, OPENINFERENCE_SPAN_KIND, OTEL_AGENT_EVAL_SCOPE, OUTPUT_VALUE, OtlpFileTraceStore, PairwiseSteeringOptimizer, ProductClient, PromptRegistry, REDACTION_VERSION, REFERENCE_EQUIVALENCE_INPUT_LIMITS, REFERENCE_EQUIVALENCE_JUDGE_VERSION, RESEARCH_REPORT_HARD_PAIR_FLOOR, ROLLOUT_SCHEMA, RUN_COST_ATTR_KEYS, ReplayCache, ReplayCacheMissError, ReplayError, RunCritic, RunIntegrityError, RunRecordValidationError, SEMANTIC_CONCEPT_JUDGE_VERSION, SKILL_USAGE_ANALYST, SPAN_KIND_ATTR_KEYS, SUPERVISOR_RUN_SCHEMA, SandboxHarness, ScenarioRegistry, SeatUnsetError, SingleBackendError, SkillUsageAnalyst, SpanNotFoundError, SubprocessSandboxDriver, TOOL_ARGS_CAPTURED, TOOL_LATENCY_MS, TOOL_NAME, TOOL_NAME_ATTR_KEYS, TRACE_ANALYST_ACTOR_DESCRIPTION, TRACE_ANALYST_ACTOR_DESCRIPTION_VERSION, TRACE_ANALYST_TRUNCATION_MARKER_PREFIX, TRACE_SCHEMA_VERSION, TokenCounter, TraceContractBuilder, TraceEmitter, TraceFileMissingError, TraceNotFoundError, UI_FINDING_SEVERITIES, UI_LENSES, UNIVERSAL_FINDERS, ValidationError, VerificationError, WILCOXON_EXACT_MAX_N, acquisitionPlansForKnowledgeGaps, agentProfileCellHashMaterial, agentProfileCellKey, agentProfileHash, agentProfileId, agentProfileModelId, agentVisibleFields, aggregateJudgeVerdicts, aggregateLlm, aggregatePrReviewScore, aggregateRunScore, allCriticalPassed, analyzeAntiSlop, analyzeSeries, analyzeSupervisorRun, analyzeSupervisorRunSources, analyzeTraces, appendScorecard, applyLlmSpanOtlpAttributes, applyToolSpanOtlpAttributes, argHash, asNumber, asString, assertCapabilityHeadroom, assertCrossFamily, assertLlmRoute, assertMinted, assertMintedLines, assertModelsServed, assertNoHiddenLeak, assertProductBenchmarkRun, assertRealAgentReceipts, assertRealBackend, assertReleaseConfidence, assertRolloutLine, assertRunAgentProfileCell, assertRunCaptured, assertSingleBackend, assignFeedbackSplit, assignHeldOutTag, attachCostToReport, attest, attributeCounterfactuals, backoffMs, deterministicSplit as benchmarkDeterministicSplit, benchmarks_exports as benchmarks, benjaminiHochberg, bisect, blendHeldout, blockingKnowledgeEval, bonferroni, bootstrapCi, buildAgentInterfaceProfileCell, buildAgentProfileCell, buildDefaultAnalystRegistry, buildDriverSystemPrompt, buildProductBenchmarkManifest, buildReflectionPrompt, buildReviewerPrompt, buildTraceAnalystTools, buildTraceInsightContext, buildTraceInsightPrompt, buildTrajectory, buildWorkerDriverSystemPrompt, byteLengthRange, cachedJudge, calibrateJudge, calibrateJudgeContinuous, callLlm, callLlmJson, canaryLeakView, canonicalJson, canonicalize, capabilityHeadroom, captureFetchToRawSink, causalAttribution, checkBehavioralCanary, checkCanaries, checkSlos, checkTraceContracts, clamp01, classifyFailure, classifyOtlpSpanRole, classifyTreatment, claudeCodeSupervisorRunReader, cliffsDelta, clusteredPairedBinary, cohensD, collectionPreserved, commentsForSource, commitBisect, comparePairedArms, compareReferenceReplay, compareToBaseline, compilerJudge, completionVerdict, composeParsers, composeValidators, computeExperimentStats, computeFindingId, computeToolUseMetrics, computeTraceMetrics, confidenceInterval, containsAll, contentHash, contextInputTokens, continuousAgreement, contractJudge, controlFailureClassFromVerification, controlRunToFeedbackTrajectory, controlRunToRunRecord, convertTraceStoresToOtlp, corpusInterRaterAgreement, corpusInterRaterAgreementFromJudgeScores, costForTokenPricing, costForUsage, costReceiptFromLlm, costReceiptFromLlmError, costReport, createAnalystAi, createAntiSlopJudge, createChatClient, createDefaultReviewer, createFeedbackTrajectory, createIntentMatchJudge, createLlmCorrectnessChecker, createLlmReviewer, createOtelExporter, createOtelTracingStore, createReferenceEquivalenceJudge, createReplayFetch, createSandboxPool, createSemanticConceptJudge, createTokenRecallChecker, createTraceAnalystKind, crossTraceDiff, crowdingDistance, dataDescriptionBits, decideNextUserTurn, decideReferenceReplayPromotion, decideReferenceReplayRunPromotion, defaultBlendWeights, defaultIsMaterial, defaultProviderRedactor, defaultReferenceReplayMatcher, defaultTraceInsightPanel, deployGateLayer, describeTraceInsightScope, diffFindings, diffScorecard, discoverPersonas, distillPlaybook, domainEvidencePattern, dominates, eProcess, ensembleJudge, errorStreakDetector, estimateCost, estimateTokens, evaluateActionPolicy, evaluateContract, evaluateHypothesis, evaluateInterimReleaseConfidence, evaluateOracles, evaluateReleaseConfidence, evaluateTraceContract, executeScenario, expandProfileAxes, expectAgent, exportProductBenchmark, exportProductBenchmarkRuns, exportRewardModel, exportRunAsOtlp, extractAssetUrls, extractErrorCount, extractOtlpAttributes, extractProducedState, extractUsage, extractUsageFromResponse, extractUsageFromSse, feedbackTrajectoriesToDatasetScenarios, feedbackTrajectoriesToOptimizerRows, feedbackTrajectoryToDatasetScenario, feedbackTrajectoryToOptimizerRow, fileContains, fileExists, fileExperimentStore, fileVerdictCache, findAutoMatchNoExpectation, findConstructorCwdDropped, findFallbackToPass, findLiteralTruePass, findProductBenchmarkArtifacts, findSkipCountsAsPass, firstNumberAttr, firstStringAttr, flattenOtlpExportToNdjson, flowLayer, fnv1a32, formatBenchmarkReport, formatDriverReport, formatFindings, formatScorecardDiff, fromHarborTrajectory, gainHistogram, gateTreatmentApplied, gateTreatmentFromMetrics, gateTreatmentFromSpans, gateTreatmentFromToolSpans, ghCliClient, gitProvenanceReader, precision as goldenPrecision, gradeOnHidden, gradeSemanticStatus, groupBy, groupRunsByAgentProfileCell, harnessAxisOf, hasCapturedToolArgs, hashContent, hashJson, hashScenarios, hashToUnit, hiddenGrade, holm, htmlContainsElement, httpGithubClient, improvementVerdict, inMemoryExperimentStore, inMemoryReferenceReplayStore, inMemoryReviewStore, inMemoryRunRecordBackend, inMemoryVerdictCache, inferDomainKeywords, inferOtlpKind, interRaterReliability, interpretCliffs, iqr, isHiddenDestination, isJudgeSpan, isLlmSpan, isModelPriced, isOtelConfigured, isOtlpModelCall, isRealnessGated, isRetrievalSpan, isRolloutLine, isRunRecord, isSandboxSpan, isToolSpan, isTrainableSplit, isTransientLlmError, isUnavailable, iterateRawCalls, jestTestParser, jsonHasKeys, jsonShape, jsonlReferenceReplayStore, jsonlReviewStore, jsonlRunRecordBackend, judgeFamily, judgeReplayGate, judgeSpans, keyPreserved, knowledgeReadinessTracePayload, leaderboard, linterJudge, llmJudge, llmSpanFromProvider, llmSpans, loadScorecard, loadScorerFromGrader, localCommandRunner, lowercaseMutator, makeEvalTools, makeFinding, mannWhitneyU, mapConcurrent, matchGoldens, matchSpan, maximumChargeForLlmRequest, mcnemar, mcnemarPower, mcnemarRequiredN, mergeLayerResults, mergeSteeringBundle, minimumPairsForPairedDeltaTest, mintRolloutRows, modelDescriptionBits, modelHasSnapshot, modelPriceKey, mulberry32, multiToolchainLayer, noProgressDetector, normalCdf, normalizeScores, notBlocked, objectiveEval, observeAll, observedScore, observedSplitScore, otelRunCompleteHook, otlpRowsToRunRecords, otlpRowsToTraceRunRecords, otlpToRunRecords, otlpToTraceRunRecords, pairArms, pairRunRecords, pairedBootstrap, pairedCohensDz, pairedDeltaTest, pairedEvalueSequence, pairedMde, pairedRiskDifference, pairedSignTest, pairedTTest, paraphraseRobustness, paraphraseRobustnessScenarios, paretoChart, paretoFrontier, paretoFrontierWithCrowding, parseCorrectnessResponse, parseFeedbackTrajectoriesJsonl, parseReflectionResponse, parseRunRecordSafe, parseRuntimeTrajectoryHookEvent, partialCredit, partitionHeldOut, passAtK, passOrthogonality, pearsonR, pixelDeltaRatio, planTraceInsightQuestions, politenessPrefixMutator, positionalBias, preflightModels, printDriverSummary, probeLlm, productBenchmarkIntegrityFailures, productBenchmarkMutableSurfaces, productBenchmarkRepoIdentity, productBenchmarkSplits, profile_exports as profile, projectOtlpFlatLine, projectRuntimeTrajectoryEvidence, promptBisect, proposeSynthesisTargets, providerFromBaseUrl, pytestTestParser, ranks, readClaudeCodeSupervisorRun, readOtlpStatus, readProductBenchmarkManifest, readProductBenchmarkRecords, recordRuns, recordRunsToScorecard, redTeamDataset, redTeamReport, redactString, redactValue, referenceReplayRunsToSteeringRows, referenceReplayScenarioToRunScore, regexMatch, regexMatches, relabelImportedSplit, renderMarkdownReport, renderPlaybookMarkdown, renderPreferenceMemoryMarkdown, renderPriorFindings, renderReleaseReport, renderSteeringText, renderSupervisorRunHeadline, renderSupervisorRunMarkdown, renderUpstreamFindings, repeatedActionDetector, replayFeedbackTrajectories, replayFeedbackTrajectory, replayScorerOverCorpus, replayTraceThroughJudge, requireAgentProfileCell, requiredPairedSampleSize, requiredSampleSize, researchReport, resolveModelPricing, resolveSeat, rollupSupervisorRuns, roundTripRunRecord, routeFields, rowCount, rowWhere, runAgentControlLoop, runAssertions, runBehavioralCanaries, runCanaries, runCounterfactual, runE2EWorkflow, runEvalCampaign, runExpectations, runFailureClass, runHarnessExperiment, runIntentMatchJudge, runJudgeFleet, runKeywordCoverageJudge, runKeywordCoverageJudgeUrl, runLiveProof, runProposeReview, runProposeReviewAsControlLoop, runRecordToProductBenchmarkRecord, runReferenceEquivalenceJudge, runReferenceReplay, runScore, runSelfPlay, runSemanticConceptJudge, runTaskScore, runTestGradedScenario, runsForScenario, scalarScore, scanForMuffledGates, scoreContinuity, scoreFromEvals, scoreKnowledgeReadiness, scoreOrigin, scorePrReviewComments, scorePrReviewSource, scoreRedTeamOutput, scoreReferenceReplay, scoreTraceInsightReadiness, seatPresets, securityJudge, selectHarnessVariant, selfPreference, sentenceReorderMutator, serializeFeedbackTrajectoriesJsonl, showMeasured, signManifest, spearmanR, statusAdvanced, stopOnNoProgress, stopOnRepeatedAction, stringField, stripFencedJson, studentTCdf, subjectiveEval, summarizeAgentReceiptIntegrity, summarizeBackendIntegrity, summarizeHarnessResults, summarizePrReviewBenchmark, summarizePreferenceMemory, summaryTable, supervisorRunRolloutLines, testJudge, textInSnapshot, throwIfRunIncomplete, toAgentProfileJson, toHarborTrajectories, toHarborTrajectory, toJsonl, toLangfuseEnvelope, toOpenAiTool, toPrometheusText, toRewardRows, toSftRows, tokenizeDomainWords, toolNamesForRun, toolSpans, traceAnalystFunctionGroup, traceAnalystOnRunComplete, traceContract, traceJudge, traceJudgeEnsemble, traceSpanKindToOpenInferenceKind, tracedAnalyzeTraces, trainingReward, trainingScore, typoMutator, urlContains, userQuestionsForKnowledgeGaps, validateAgentProfileCell, validateProductBenchmarkManifest, validateProductBenchmarkRecord, validateProductBenchmarkRun, validateRolloutLine, validateRunRecord, verbosityBias, verifyAgentProfileCell, verifyAttestation, verifyCompletion, verifyManifest, visualDiff, viteDeployRunner, vitestTestParser, weightedComposite, weightedMean, weightedRecall, welchsTTest, whitespaceCollapseMutator, wilcoxonSignedRank, wilson, withAssignedFeedbackSplit, withHeldoutBlend, withJudgeRetry, withOtelPipeline, wranglerDeployRunner, writeSupervisorRunReport };
12296
+ export { AGENT_PROFILE_KINDS, ATIF_SCHEMA_VERSION, ATTESTATION_ALGORITHM, AgentDriver, AgentEvalError, AgentProfileCellValidationError, AnalystRegistry, AxGepaSteeringOptimizer, BENCHMARK_SPLIT_SEED, BOOTSTRAP_GATE_MIN_N, BackendIntegrityError, BenchmarkRunner, BudgetBreachError, BudgetGuard, CODING_HARNESSES, CallbackResearcher, CaptureIntegrityError, ConfigError, ConvergenceTracker, CostAccountingIncompleteError, CostCallConflictError, CostCeilingReachedError, CostLedger, CostLedgerPersistenceError, CostReceiptCaptureError, CostReservationExceededError, CostTracker, CrossFamilyError, DEFAULT_AGENT_SLOS, DEFAULT_COMPLEXITY_WEIGHTS, DEFAULT_RULES as DEFAULT_FAILURE_RULES, DEFAULT_FINDERS, DEFAULT_HARNESS_OBJECTIVES, DEFAULT_MUTATION_PRIMITIVES, DEFAULT_MUTATORS, DEFAULT_PERMUTATIONS, DEFAULT_PR_REVIEW_SCORE_WEIGHTS, DEFAULT_REDACTION_RULES, DEFAULT_RED_TEAM_CORPUS, DEFAULT_RUN_SCORE_WEIGHTS, DEFAULT_SEVERITY_WEIGHTS, DEFAULT_TRACE_ANALYST_BUDGETS, DEFAULT_TRACE_ANALYST_KINDS, Dataset, DescriptionLengthGate, DockerSandboxDriver, DualAgentBench, ERROR_COUNT_PATTERNS, EvalTraceStore, ExperimentTracker, FAILURE_CLASSES, FAILURE_MODE_KIND_SPEC, FileSystemFeedbackTrajectoryStore, FileSystemRawProviderSink, FileSystemTraceStore, FindingsStore, HARBOR_IMPORT_GAP, HARNESS_NATIVE_MODEL, HeldOutGate, HoldoutAuditor, HoldoutLockedError, IMPROVEMENT_KIND_SPEC, INPUT_VALUE, INTENT_MATCH_JUDGE_VERSION, InMemoryFeedbackTrajectoryStore, InMemoryRawProviderSink, InMemoryTraceStore, InMemoryWorkspaceInspector, JudgeError, JudgeParseError, JudgeRunner, KNOWLEDGE_GAP_KIND_SPEC, KNOWLEDGE_POISONING_KIND_SPEC, LLM_CACHED_TOKENS, LLM_CACHED_TOKEN_ATTR_KEYS, LLM_CACHE_WRITE_TOKENS, LLM_CACHE_WRITE_TOKEN_ATTR_KEYS, LLM_CONTEXT_TOKENS, LLM_COST_ATTR_KEYS, LLM_COST_USD, LLM_INPUT_TOKENS, LLM_INPUT_TOKEN_ATTR_KEYS, LLM_MODEL_ATTR_KEYS, LLM_MODEL_NAME, LLM_OUTPUT_TOKENS, LLM_OUTPUT_TOKEN_ATTR_KEYS, LLM_REASONING_TOKENS, LLM_REASONING_TOKEN_ATTR_KEYS, LlmCallError, LlmClient, LlmResponseError, LlmRouteAssertionError, LockedJsonlAppender, MANN_WHITNEY_EXACT_MAX_STATES, MANN_WHITNEY_EXACT_MAX_WORK, MODEL_PRICING, MetricsCollector, ModelsUnreachableError, MultiLayerVerifier, Mutex, NoopRawProviderSink, NoopResearcher, NotFoundError, OPENINFERENCE_SPAN_KIND, OTEL_AGENT_EVAL_SCOPE, OUTPUT_VALUE, OtlpFileTraceStore, PairwiseSteeringOptimizer, ProductClient, PromptRegistry, REDACTION_VERSION, REFERENCE_EQUIVALENCE_INPUT_LIMITS, REFERENCE_EQUIVALENCE_JUDGE_VERSION, RESEARCH_REPORT_HARD_PAIR_FLOOR, ROLLOUT_SCHEMA, RUN_COST_ATTR_KEYS, ReplayCache, ReplayCacheMissError, ReplayError, RunCritic, RunIntegrityError, RunRecordValidationError, SEMANTIC_CONCEPT_JUDGE_VERSION, SKILL_USAGE_ANALYST, SPAN_KIND_ATTR_KEYS, SUPERVISOR_RUN_SCHEMA, SandboxHarness, ScenarioRegistry, SeatUnsetError, SingleBackendError, SkillUsageAnalyst, SpanNotFoundError, SubprocessSandboxDriver, TOOL_ARGS_CAPTURED, TOOL_LATENCY_MS, TOOL_NAME, TOOL_NAME_ATTR_KEYS, TRACE_ANALYST_ACTOR_DESCRIPTION, TRACE_ANALYST_ACTOR_DESCRIPTION_VERSION, TRACE_ANALYST_TRUNCATION_MARKER_PREFIX, TRACE_SCHEMA_VERSION, TokenCounter, TraceContractBuilder, TraceEmitter, TraceFileMissingError, TraceNotFoundError, UI_FINDING_SEVERITIES, UI_LENSES, UNIVERSAL_FINDERS, ValidationError, VerificationError, WILCOXON_EXACT_MAX_N, acquisitionPlansForKnowledgeGaps, agentProfileCellHashMaterial, agentProfileCellKey, agentProfileHash, agentProfileId, agentProfileModelId, agentVisibleFields, aggregateJudgeVerdicts, aggregateLlm, aggregatePrReviewScore, aggregateRunScore, allCriticalPassed, analyzeAntiSlop, analyzeSeries, analyzeSupervisorRun, analyzeSupervisorRunSources, analyzeTraces, appendScorecard, applyLlmSpanOtlpAttributes, applyToolSpanOtlpAttributes, argHash, asNumber, asString, assertCapabilityHeadroom, assertCrossFamily, assertLlmRoute, assertMinted, assertMintedLines, assertModelsServed, assertNoHiddenLeak, assertProductBenchmarkRun, assertRealAgentReceipts, assertRealBackend, assertReleaseConfidence, assertRolloutLine, assertRunAgentProfileCell, assertRunCaptured, assertSingleBackend, assignFeedbackSplit, assignHeldOutTag, attachCostToReport, attest, attributeCounterfactuals, backoffMs, deterministicSplit as benchmarkDeterministicSplit, benchmarks_exports as benchmarks, benjaminiHochberg, bisect, blendHeldout, blockingKnowledgeEval, bonferroni, bootstrapCi, buildAgentInterfaceProfileCell, buildAgentProfileCell, buildDefaultAnalystRegistry, buildDriverSystemPrompt, buildProductBenchmarkManifest, buildReflectionPrompt, buildReviewerPrompt, buildTraceAnalystTools, buildTraceInsightContext, buildTraceInsightPrompt, buildTrajectory, buildWorkerDriverSystemPrompt, byteLengthRange, cachedJudge, calibrateJudge, calibrateJudgeContinuous, callLlm, callLlmJson, canaryLeakView, canonicalJson, canonicalize, capabilityHeadroom, captureFetchToRawSink, causalAttribution, checkBehavioralCanary, checkCanaries, checkSlos, checkTraceContracts, clamp01, classifyFailure, classifyOtlpSpanRole, classifyTreatment, claudeCodeSupervisorRunReader, cliffsDelta, clusteredPairedBinary, cohensD, collectionPreserved, commentsForSource, commitBisect, comparePairedArms, compareReferenceReplay, compareToBaseline, compilerJudge, completionVerdict, composeParsers, composeValidators, computeExperimentStats, computeFindingId, computeToolUseMetrics, computeTraceMetrics, confidenceInterval, containsAll, contentHash, contextInputTokens, continuousAgreement, contractJudge, controlFailureClassFromVerification, controlRunToFeedbackTrajectory, controlRunToRunRecord, convertTraceStoresToOtlp, corpusInterRaterAgreement, corpusInterRaterAgreementFromJudgeScores, costForTokenPricing, costForUsage, costReceiptFromLlm, costReceiptFromLlmError, costReport, createAnalystAi, createAntiSlopJudge, createChatClient, createDefaultReviewer, createFeedbackTrajectory, createIntentMatchJudge, createLlmCorrectnessChecker, createLlmReviewer, createOtelExporter, createOtelTracingStore, createReferenceEquivalenceJudge, createReplayFetch, createSandboxPool, createSemanticConceptJudge, createTokenRecallChecker, createTraceAnalystKind, crossTraceDiff, crowdingDistance, dataDescriptionBits, decideNextUserTurn, decideReferenceReplayPromotion, decideReferenceReplayRunPromotion, defaultBlendWeights, defaultIsMaterial, defaultProviderRedactor, defaultReferenceReplayMatcher, defaultTraceInsightPanel, deployGateLayer, describeTraceInsightScope, diffFindings, diffScorecard, discoverPersonas, distillPlaybook, domainEvidencePattern, dominates, eProcess, ensembleJudge, errorStreakDetector, estimateCost, estimateTokens, evaluateActionPolicy, evaluateContract, evaluateHypothesis, evaluateInterimReleaseConfidence, evaluateOracles, evaluateReleaseConfidence, evaluateTraceContract, executeScenario, expandProfileAxes, expectAgent, exportProductBenchmark, exportProductBenchmarkRuns, exportRewardModel, exportRunAsOtlp, extractAssetUrls, extractErrorCount, extractOtlpAttributes, extractProducedState, extractUsage, extractUsageFromResponse, extractUsageFromSse, feedbackTrajectoriesToDatasetScenarios, feedbackTrajectoriesToOptimizerRows, feedbackTrajectoryToDatasetScenario, feedbackTrajectoryToOptimizerRow, fileContains, fileExists, fileExperimentStore, fileVerdictCache, findAutoMatchNoExpectation, findConstructorCwdDropped, findFallbackToPass, findLiteralTruePass, findProductBenchmarkArtifacts, findSkipCountsAsPass, firstNumberAttr, firstStringAttr, flattenOtlpExportToNdjson, flowLayer, fnv1a32, formatBenchmarkReport, formatDriverReport, formatFindings, formatScorecardDiff, fromHarborTrajectory, gainHistogram, gateTreatmentApplied, gateTreatmentFromMetrics, gateTreatmentFromSpans, gateTreatmentFromToolSpans, ghCliClient, gitProvenanceReader, precision as goldenPrecision, gradeOnHidden, gradeSemanticStatus, groupBy, groupRunsByAgentProfileCell, harnessAxisOf, hasCapturedToolArgs, hashContent, hashJson, hashScenarios, hashToUnit, hiddenGrade, holm, htmlContainsElement, httpGithubClient, improvementVerdict, inMemoryExperimentStore, inMemoryReferenceReplayStore, inMemoryReviewStore, inMemoryRunRecordBackend, inMemoryVerdictCache, inferDomainKeywords, inferOtlpKind, interRaterReliability, interpretCliffs, iqr, isHiddenDestination, isJudgeSpan, isLlmSpan, isModelPriced, isOtelConfigured, isOtlpModelCall, isRealnessGated, isRetrievalSpan, isRolloutLine, isRunRecord, isSandboxSpan, isToolSpan, isTrainableSplit, isTransientLlmError, isUnavailable, iterateRawCalls, jestTestParser, jsonHasKeys, jsonShape, jsonlReferenceReplayStore, jsonlReviewStore, jsonlRunRecordBackend, judgeFamily, judgeReplayGate, judgeSpans, keyPreserved, knowledgeReadinessTracePayload, leaderboard, linterJudge, llmJudge, llmSpanFromProvider, llmSpans, loadScorecard, loadScorerFromGrader, localCommandRunner, lowercaseMutator, makeEvalTools, makeFinding, makeProposalFinding, mannWhitneyU, mapConcurrent, matchGoldens, matchSpan, maximumChargeForLlmRequest, mcnemar, mcnemarPower, mcnemarRequiredN, mergeLayerResults, mergeSteeringBundle, minimumPairsForPairedDeltaTest, mintRolloutRows, modelDescriptionBits, modelHasSnapshot, modelPriceKey, mulberry32, multiToolchainLayer, noProgressDetector, normalCdf, normalizeScores, notBlocked, objectiveEval, observeAll, observedScore, observedSplitScore, otelRunCompleteHook, otlpRowsToRunRecords, otlpRowsToTraceRunRecords, otlpToRunRecords, otlpToTraceRunRecords, pairArms, pairRunRecords, pairedBootstrap, pairedCohensDz, pairedDeltaTest, pairedEvalueSequence, pairedMde, pairedRiskDifference, pairedSignTest, pairedTTest, paraphraseRobustness, paraphraseRobustnessScenarios, paretoChart, paretoFrontier, paretoFrontierWithCrowding, parseCorrectnessResponse, parseFeedbackTrajectoriesJsonl, parseReflectionResponse, parseRunRecordSafe, parseRuntimeTrajectoryHookEvent, partialCredit, partitionHeldOut, passAtK, passOrthogonality, pearsonR, pixelDeltaRatio, planTraceInsightQuestions, politenessPrefixMutator, positionalBias, preflightModels, printDriverSummary, probeLlm, productBenchmarkIntegrityFailures, productBenchmarkMutableSurfaces, productBenchmarkRepoIdentity, productBenchmarkSplits, profile_exports as profile, projectOtlpFlatLine, projectRuntimeTrajectoryEvidence, promptBisect, proposeSynthesisTargets, providerFromBaseUrl, pytestTestParser, ranks, readClaudeCodeSupervisorRun, readOtlpStatus, readProductBenchmarkManifest, readProductBenchmarkRecords, recordRuns, recordRunsToScorecard, redTeamDataset, redTeamReport, redactString, redactValue, referenceReplayRunsToSteeringRows, referenceReplayScenarioToRunScore, regexMatch, regexMatches, relabelImportedSplit, renderMarkdownReport, renderPlaybookMarkdown, renderPreferenceMemoryMarkdown, renderPriorFindings, renderReleaseReport, renderSteeringText, renderSupervisorRunHeadline, renderSupervisorRunMarkdown, renderUpstreamFindings, repeatedActionDetector, replayFeedbackTrajectories, replayFeedbackTrajectory, replayScorerOverCorpus, replayTraceThroughJudge, requireAgentProfileCell, requiredPairedSampleSize, requiredSampleSize, researchReport, resolveModelPricing, resolveSeat, rollupSupervisorRuns, roundTripRunRecord, routeFields, rowCount, rowWhere, runAgentControlLoop, runAssertions, runBehavioralCanaries, runCanaries, runCounterfactual, runE2EWorkflow, runEvalCampaign, runExpectations, runFailureClass, runHarnessExperiment, runIntentMatchJudge, runJudgeFleet, runKeywordCoverageJudge, runKeywordCoverageJudgeUrl, runLiveProof, runProposeReview, runProposeReviewAsControlLoop, runRecordToProductBenchmarkRecord, runReferenceEquivalenceJudge, runReferenceReplay, runScore, runSelfPlay, runSemanticConceptJudge, runTaskScore, runTestGradedScenario, runsForScenario, scalarScore, scanForMuffledGates, scoreContinuity, scoreFromEvals, scoreKnowledgeReadiness, scoreOrigin, scorePrReviewComments, scorePrReviewSource, scoreRedTeamOutput, scoreReferenceReplay, scoreTraceInsightReadiness, seatPresets, securityJudge, selectHarnessVariant, selfPreference, sentenceReorderMutator, serializeFeedbackTrajectoriesJsonl, showMeasured, signManifest, spearmanR, statusAdvanced, stopOnNoProgress, stopOnRepeatedAction, stringField, stripFencedJson, studentTCdf, subjectiveEval, summarizeAgentReceiptIntegrity, summarizeBackendIntegrity, summarizeHarnessResults, summarizePrReviewBenchmark, summarizePreferenceMemory, summaryTable, supervisorRunRolloutLines, testJudge, textInSnapshot, throwIfRunIncomplete, toAgentProfileJson, toHarborTrajectories, toHarborTrajectory, toJsonl, toLangfuseEnvelope, toOpenAiTool, toPrometheusText, toRewardRows, toSftRows, tokenizeDomainWords, toolNamesForRun, toolSpans, traceAnalystFunctionGroup, traceAnalystOnRunComplete, traceContract, traceJudge, traceJudgeEnsemble, traceSpanKindToOpenInferenceKind, tracedAnalyzeTraces, trainingReward, trainingScore, typoMutator, urlContains, userQuestionsForKnowledgeGaps, validateAgentProfileCell, validateProductBenchmarkManifest, validateProductBenchmarkRecord, validateProductBenchmarkRun, validateRolloutLine, validateRunRecord, verbosityBias, verifyAgentProfileCell, verifyAttestation, verifyCompletion, verifyManifest, visualDiff, viteDeployRunner, vitestTestParser, weightedComposite, weightedMean, weightedRecall, welchsTTest, whitespaceCollapseMutator, wilcoxonSignedRank, wilson, withAssignedFeedbackSplit, withHeldoutBlend, withJudgeRetry, withOtelPipeline, wranglerDeployRunner, writeSupervisorRunReport };
12127
12297
 
12128
12298
  //# sourceMappingURL=index.js.map