@tangle-network/agent-eval 0.133.3 → 0.134.1
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +70 -0
- package/dist/analyst/index.d.ts +11 -35
- package/dist/analyst/index.d.ts.map +1 -1
- package/dist/analyst/index.js +4 -53
- package/dist/analyst/index.js.map +1 -1
- package/dist/{analyze-runs-BClW9OSe.d.ts → analyze-runs-DMo3Lb_y.d.ts} +4 -4
- package/dist/{analyze-runs-BClW9OSe.d.ts.map → analyze-runs-DMo3Lb_y.d.ts.map} +1 -1
- package/dist/benchmarks/index.d.ts +1 -1
- package/dist/benchmarks/index.js +1 -1
- package/dist/{benchmarks-BP9sgMia.js → benchmarks-v5piCeDl.js} +3 -3
- package/dist/{benchmarks-BP9sgMia.js.map → benchmarks-v5piCeDl.js.map} +1 -1
- package/dist/campaign/index.d.ts +5 -4
- package/dist/campaign/index.js +4 -3
- package/dist/{campaign--V4ffEKR.js → campaign-DEC_7DLn.js} +2 -2
- package/dist/{campaign--V4ffEKR.js.map → campaign-DEC_7DLn.js.map} +1 -1
- package/dist/{client-Du7B81wW.d.ts → client-BIyh1RCr.d.ts} +3 -3
- package/dist/{client-Du7B81wW.d.ts.map → client-BIyh1RCr.d.ts.map} +1 -1
- package/dist/contract/index.d.ts +12 -11
- package/dist/contract/index.d.ts.map +1 -1
- package/dist/contract/index.js +4 -11
- package/dist/contract/index.js.map +1 -1
- package/dist/{default-registry-Cl3pHo4n.d.ts → default-registry-Brxr728w.d.ts} +4 -268
- package/dist/default-registry-Brxr728w.d.ts.map +1 -0
- package/dist/{default-registry-D3T9XbuY.js → default-registry-IjYs7T8l.js} +4 -61
- package/dist/default-registry-IjYs7T8l.js.map +1 -0
- package/dist/hosted/index.d.ts +2 -2
- package/dist/{index-DOqvIJ8I.d.ts → index-BoJNQR6n.d.ts} +4 -3
- package/dist/index-BoJNQR6n.d.ts.map +1 -0
- package/dist/{index-B5MNN1f1.d.ts → index-C21xKtxu.d.ts} +4 -4
- package/dist/{index-B5MNN1f1.d.ts.map → index-C21xKtxu.d.ts.map} +1 -1
- package/dist/index.d.ts +13 -12
- package/dist/index.d.ts.map +1 -1
- package/dist/index.js +190 -20
- package/dist/index.js.map +1 -1
- package/dist/multishot/index.d.ts +1 -1
- package/dist/openapi.json +1 -1
- package/dist/proposal-findings-DCawte-y.js +164 -0
- package/dist/proposal-findings-DCawte-y.js.map +1 -0
- package/dist/{release-report-DKBtegGt.d.ts → release-report-CuULWKyk.d.ts} +2 -2
- package/dist/{release-report-DKBtegGt.d.ts.map → release-report-CuULWKyk.d.ts.map} +1 -1
- package/dist/reporting.d.ts +2 -2
- package/dist/{researcher-BtD5U1Up.d.ts → researcher-DVtruQ9U.d.ts} +2 -2
- package/dist/{researcher-BtD5U1Up.d.ts.map → researcher-DVtruQ9U.d.ts.map} +1 -1
- package/dist/rl.d.ts +2 -2
- package/dist/rl.d.ts.map +1 -1
- package/dist/rl.js +14 -1
- package/dist/rl.js.map +1 -1
- package/dist/{semantic-concept-judge-BypLt6Fw.js → semantic-concept-judge-C0P1VTXD.js} +2 -3
- package/dist/{semantic-concept-judge-BypLt6Fw.js.map → semantic-concept-judge-C0P1VTXD.js.map} +1 -1
- package/dist/{skill-usage-BaaxFSJR.d.ts → skill-usage-BDQVPIG1.d.ts} +3 -2
- package/dist/skill-usage-BDQVPIG1.d.ts.map +1 -0
- package/dist/{skillopt-optimization-method-vvJ4bMNI.js → skillopt-optimization-method-BY6vKLJB.js} +51 -30
- package/dist/skillopt-optimization-method-BY6vKLJB.js.map +1 -0
- package/dist/{skillopt-optimization-method-Dxr8pdZd.d.ts → skillopt-optimization-method-DJ3l4w8W.d.ts} +10 -13
- package/dist/skillopt-optimization-method-DJ3l4w8W.d.ts.map +1 -0
- package/dist/{summary-report-DyOhItws.d.ts → summary-report-DGp0-_XO.d.ts} +59 -3
- package/dist/summary-report-DGp0-_XO.d.ts.map +1 -0
- package/dist/types-DVjczBM9.d.ts +276 -0
- package/dist/types-DVjczBM9.d.ts.map +1 -0
- package/dist/{types-BokuXvOG.d.ts → types-DiWLru6Z.d.ts} +20 -37
- package/dist/types-DiWLru6Z.d.ts.map +1 -0
- package/docs/campaign-proposers.md +5 -0
- package/package.json +1 -1
- package/dist/default-registry-Cl3pHo4n.d.ts.map +0 -1
- package/dist/default-registry-D3T9XbuY.js.map +0 -1
- package/dist/index-DOqvIJ8I.d.ts.map +0 -1
- package/dist/run-score-iEEAWiBY.js +0 -41
- package/dist/run-score-iEEAWiBY.js.map +0 -1
- package/dist/skill-usage-BaaxFSJR.d.ts.map +0 -1
- package/dist/skillopt-optimization-method-Dxr8pdZd.d.ts.map +0 -1
- package/dist/skillopt-optimization-method-vvJ4bMNI.js.map +0 -1
- package/dist/summary-report-DyOhItws.d.ts.map +0 -1
- package/dist/types-BokuXvOG.d.ts.map +0 -1
package/dist/index.js
CHANGED
|
@@ -2,16 +2,16 @@ import { t as __exportAll } from "./rolldown-runtime-8H4AJuhK.js";
|
|
|
2
2
|
import { _ as observeAll, a as jsonlReviewStore, c as scoreFromEvals, d as runAgentControlLoop, f as stopOnNoProgress, g as noProgressDetector, h as errorStreakDetector, i as inMemoryReviewStore, l as allCriticalPassed, m as subjectiveEval, n as runProposeReviewAsControlLoop, o as runProposeReview, p as stopOnRepeatedAction, r as createLlmReviewer, s as controlRunToRunRecord, t as controlFailureClassFromVerification, u as objectiveEval, v as repeatedActionDetector, y as evaluateActionPolicy } from "./propose-review-control-SQ-n9-We.js";
|
|
3
3
|
import { a as NotFoundError, c as VerificationError, i as JudgeError, n as CaptureIntegrityError, o as ReplayError, r as ConfigError, s as ValidationError, t as AgentEvalError } from "./errors-8YnH8WlF.js";
|
|
4
4
|
import { _ as verifyManifest, a as assertRunAgentProfileCell, c as groupRunsByAgentProfileCell, d as validateAgentProfileCell, f as verifyAgentProfileCell, g as signManifest, h as hashJson, i as agentProfileCellKey, l as requireAgentProfileCell, m as evaluateHypothesis, n as AgentProfileCellValidationError, o as buildAgentInterfaceProfileCell, p as canonicalize, r as agentProfileCellHashMaterial, s as buildAgentProfileCell, t as AGENT_PROFILE_KINDS, u as toAgentProfileJson } from "./agent-profile-cell-OhuTee9n.js";
|
|
5
|
-
import { F as
|
|
5
|
+
import { F as createChatClient, I as createAnalystAi, P as computeTraceMetrics, a as KNOWLEDGE_GAP_KIND_SPEC, d as renderUpstreamFindings, i as KNOWLEDGE_POISONING_KIND_SPEC, l as createTraceAnalystKind, n as AnalystRegistry, o as IMPROVEMENT_KIND_SPEC, r as DEFAULT_TRACE_ANALYST_KINDS, s as FAILURE_MODE_KIND_SPEC, t as buildDefaultAnalystRegistry, u as renderPriorFindings } from "./default-registry-IjYs7T8l.js";
|
|
6
6
|
import { _ as resolveModelPricing, a as CostLedgerPersistenceError, c as costForTokenPricing, d as MODEL_PRICING, f as MetricsCollector, g as isModelPriced, h as estimateTokens, i as CostLedger, l as costForUsage, m as estimateCost, n as CostCallConflictError, o as CostReceiptCaptureError, p as TokenCounter, r as CostCeilingReachedError, s as CostReservationExceededError, t as CostAccountingIncompleteError, u as modelPriceKey } from "./cost-ledger-BrJxbrMy.js";
|
|
7
7
|
import { a as providerFromBaseUrl, i as defaultProviderRedactor, n as InMemoryRawProviderSink, r as NoopRawProviderSink, t as FileSystemRawProviderSink } from "./raw-provider-sink-BQd7mzyT.js";
|
|
8
8
|
import { a as assertLlmRoute, c as callLlmJson, d as isTransientLlmError, f as maximumChargeForLlmRequest, i as LlmRouteAssertionError, l as costReceiptFromLlm, m as stripFencedJson, n as LlmClient, o as backoffMs, p as probeLlm, r as LlmResponseError, s as callLlm, t as LlmCallError, u as costReceiptFromLlmError } from "./llm-client-ClPW-dWB.js";
|
|
9
9
|
import { INPUT_VALUE, LLM_CACHED_TOKENS, LLM_CACHED_TOKEN_ATTR_KEYS, LLM_CACHE_WRITE_TOKENS, LLM_CACHE_WRITE_TOKEN_ATTR_KEYS, LLM_CONTEXT_TOKENS, LLM_COST_ATTR_KEYS, LLM_COST_USD, LLM_INPUT_TOKENS, LLM_INPUT_TOKEN_ATTR_KEYS, LLM_MODEL_ATTR_KEYS, LLM_MODEL_NAME, LLM_OUTPUT_TOKENS, LLM_OUTPUT_TOKEN_ATTR_KEYS, LLM_REASONING_TOKENS, LLM_REASONING_TOKEN_ATTR_KEYS, OPENINFERENCE_SPAN_KIND, OUTPUT_VALUE, RUN_COST_ATTR_KEYS, SPAN_KIND_ATTR_KEYS, TOOL_ARGS_CAPTURED, TOOL_LATENCY_MS, TOOL_NAME, TOOL_NAME_ATTR_KEYS, applyLlmSpanOtlpAttributes, asNumber, contextInputTokens, firstNumberAttr } from "./trace-attributes.js";
|
|
10
10
|
import { S as traceSpanKindToOpenInferenceKind, a as SpanNotFoundError, b as classifyOtlpSpanRole, c as DEFAULT_TRACE_ANALYST_BUDGETS, f as extractOtlpAttributes, g as readOtlpStatus, h as projectOtlpFlatLine, i as OtlpFileTraceStore, l as TRACE_ANALYST_TRUNCATION_MARKER_PREFIX, m as inferOtlpKind, n as traceAnalystFunctionGroup, o as TraceFileMissingError, p as firstStringAttr, s as TraceNotFoundError, t as buildTraceAnalystTools, u as asString, v as stringField, x as isOtlpModelCall, y as applyToolSpanOtlpAttributes } from "./tools-BmuN627J.js";
|
|
11
|
-
import {
|
|
11
|
+
import { a as clamp01, c as makeFinding, i as aggregateRunScore, l as makeProposalFinding, r as DEFAULT_RUN_SCORE_WEIGHTS, s as computeFindingId } from "./proposal-findings-DCawte-y.js";
|
|
12
12
|
import { n as mapConcurrent, t as Mutex } from "./concurrency-MUjT7VjM.js";
|
|
13
|
-
import { a as RunCritic, d as defaultIsMaterial, f as diffFindings, i as runSemanticConceptJudge, n as SEMANTIC_CONCEPT_JUDGE_VERSION, o as SKILL_USAGE_ANALYST, p as LockedJsonlAppender, r as createSemanticConceptJudge, s as SkillUsageAnalyst, t as DEFAULT_COMPLEXITY_WEIGHTS, u as FindingsStore } from "./semantic-concept-judge-
|
|
14
|
-
import { At as dominates, B as surfaceContentHash, Bt as summarizeAgentReceiptIntegrity, Ct as llmJudge, D as runCanaries, Dt as fileVerdictCache, Et as contentHash, Ft as minimumPairsForPairedDeltaTest, G as DEFAULT_MUTATION_PRIMITIVES, Ht as JudgeParseError, It as pairedDeltaTest, K as buildReflectionPrompt, Lt as BackendIntegrityError, Mt as paretoFrontierWithCrowding, Nt as scalarScore, Ot as inMemoryVerdictCache, Rt as assertRealAgentReceipts, St as hashScenarios, Tt as canonicalJson, Vt as summarizeBackendIntegrity, _t as redTeamReport, bt as Dataset, dt as REFERENCE_EQUIVALENCE_INPUT_LIMITS, ft as REFERENCE_EQUIVALENCE_JUDGE_VERSION, gt as redTeamDataset, ht as DEFAULT_RED_TEAM_CORPUS, jt as paretoFrontier, kt as crowdingDistance, mt as runReferenceEquivalenceJudge, pt as createReferenceEquivalenceJudge, q as parseReflectionResponse, vt as scoreRedTeamOutput, wt as cachedJudge, xt as HoldoutLockedError, yt as toolNamesForRun, zt as assertRealBackend } from "./skillopt-optimization-method-
|
|
13
|
+
import { a as RunCritic, d as defaultIsMaterial, f as diffFindings, i as runSemanticConceptJudge, n as SEMANTIC_CONCEPT_JUDGE_VERSION, o as SKILL_USAGE_ANALYST, p as LockedJsonlAppender, r as createSemanticConceptJudge, s as SkillUsageAnalyst, t as DEFAULT_COMPLEXITY_WEIGHTS, u as FindingsStore } from "./semantic-concept-judge-C0P1VTXD.js";
|
|
14
|
+
import { At as dominates, B as surfaceContentHash, Bt as summarizeAgentReceiptIntegrity, Ct as llmJudge, D as runCanaries, Dt as fileVerdictCache, Et as contentHash, Ft as minimumPairsForPairedDeltaTest, G as DEFAULT_MUTATION_PRIMITIVES, Ht as JudgeParseError, It as pairedDeltaTest, K as buildReflectionPrompt, Lt as BackendIntegrityError, Mt as paretoFrontierWithCrowding, Nt as scalarScore, Ot as inMemoryVerdictCache, Rt as assertRealAgentReceipts, St as hashScenarios, Tt as canonicalJson, Vt as summarizeBackendIntegrity, _t as redTeamReport, bt as Dataset, dt as REFERENCE_EQUIVALENCE_INPUT_LIMITS, ft as REFERENCE_EQUIVALENCE_JUDGE_VERSION, gt as redTeamDataset, ht as DEFAULT_RED_TEAM_CORPUS, jt as paretoFrontier, kt as crowdingDistance, mt as runReferenceEquivalenceJudge, pt as createReferenceEquivalenceJudge, q as parseReflectionResponse, vt as scoreRedTeamOutput, wt as cachedJudge, xt as HoldoutLockedError, yt as toolNamesForRun, zt as assertRealBackend } from "./skillopt-optimization-method-BY6vKLJB.js";
|
|
15
15
|
import { A as passAtK, B as studentTCdf, C as pairedBootstrap, D as pairedSignTest, E as pairedRiskDifference, F as spearmanR, G as continuousAgreement, H as normalCdf, I as weightedComposite, J as verbosityBias, K as positionalBias, L as weightedMean, M as ranks, N as requiredPairedSampleSize, O as pairedTTest, P as requiredSampleSize, R as wilcoxonSignedRank, S as normalizeScores, T as pairedMde, U as calibrateJudge, W as calibrateJudgeContinuous, _ as mannWhitneyU, a as WILCOXON_EXACT_MAX_N, b as mcnemarRequiredN, c as cliffsDelta, d as corpusInterRaterAgreement, f as corpusInterRaterAgreementFromJudgeScores, g as interpretCliffs, h as interRaterReliability, i as MANN_WHITNEY_EXACT_MAX_WORK, j as pearsonR, k as partialCredit, l as cohensD, m as holm, n as DEFAULT_PERMUTATIONS, o as benjaminiHochberg, p as eProcess, q as selfPreference, r as MANN_WHITNEY_EXACT_MAX_STATES, s as bonferroni, t as BOOTSTRAP_GATE_MIN_N, u as confidenceInterval, v as mcnemar, w as pairedCohensDz, x as mulberry32, y as mcnemarPower, z as wilson } from "./statistics-RwRNu2__.js";
|
|
16
16
|
import { n as pairArms, r as pairRunRecords, t as comparePairedArms } from "./paired-arms-CA_8pN01.js";
|
|
17
17
|
import { n as llmSpanFromProvider, t as TraceEmitter } from "./emitter-CPBAhxum.js";
|
|
@@ -28,7 +28,7 @@ import { D as showMeasured, E as isUnavailable, S as rollupSupervisorRuns, T as
|
|
|
28
28
|
import { n as TRACE_ANALYST_ACTOR_DESCRIPTION, r as TRACE_ANALYST_ACTOR_DESCRIPTION_VERSION, t as analyzeTraces } from "./analyst-LsnNpSkm.js";
|
|
29
29
|
import { C as planTraceInsightQuestions, E as traceAnalystOnRunComplete, S as inferDomainKeywords, T as tokenizeDomainWords, _ as buildTraceInsightContext, a as convertTraceStoresToOtlp, b as describeTraceInsightScope, c as otelRunCompleteHook, d as captureFetchToRawSink, f as otlpRowsToRunRecords, g as flattenOtlpExportToNdjson, h as otlpToTraceRunRecords, i as iterateRawCalls, l as OTEL_AGENT_EVAL_SCOPE, m as otlpToRunRecords, n as ReplayCacheMissError, o as createOtelExporter, p as otlpRowsToTraceRunRecords, r as createReplayFetch, s as createOtelTracingStore, t as ReplayCache, u as exportRunAsOtlp, v as buildTraceInsightPrompt, w as scoreTraceInsightReadiness, x as domainEvidencePattern, y as defaultTraceInsightPanel } from "./replay-CJfGLdx4.js";
|
|
30
30
|
import { n as extractUsageFromResponse, r as extractUsageFromSse, t as extractUsage } from "./extract-usage-2j25whHw.js";
|
|
31
|
-
import { B as harnessAxisOf, F as HARNESS_NATIVE_MODEL, G as parseCorrectnessResponse, H as completionVerdict, I as agentProfileHash, K as verifyCompletion, L as agentProfileId, P as CODING_HARNESSES, R as agentProfileModelId, U as createLlmCorrectnessChecker, V as extractProducedState, W as createTokenRecallChecker, z as expandProfileAxes } from "./campaign
|
|
31
|
+
import { B as harnessAxisOf, F as HARNESS_NATIVE_MODEL, G as parseCorrectnessResponse, H as completionVerdict, I as agentProfileHash, K as verifyCompletion, L as agentProfileId, P as CODING_HARNESSES, R as agentProfileModelId, U as createLlmCorrectnessChecker, V as extractProducedState, W as createTokenRecallChecker, z as expandProfileAxes } from "./campaign-DEC_7DLn.js";
|
|
32
32
|
import { n as iqr, r as welchsTTest, t as compareToBaseline } from "./baseline-BaPxoROc.js";
|
|
33
33
|
import { a as judgeSpans, c as runsForScenario, i as hasCapturedToolArgs, l as toolSpans, n as argHash, o as llmSpans, r as groupBy, s as runFailureClass, t as aggregateLlm } from "./query-Di7eEQ79.js";
|
|
34
34
|
import { a as checkBehavioralCanary, i as canaryLeakView, o as checkCanaries, r as HoldoutAuditor, s as runBehavioralCanaries, t as analyzeRuns } from "./analyze-runs-qk8op0tN.js";
|
|
@@ -38,7 +38,7 @@ import { a as InMemoryTraceStore, i as FileSystemTraceStore, n as assertRunCaptu
|
|
|
38
38
|
import { i as redactValue, n as REDACTION_VERSION, r as redactString, t as DEFAULT_REDACTION_RULES } from "./redact-7Aq1ukl-.js";
|
|
39
39
|
import { n as DEFAULT_RULES, r as classifyFailure, t as computeToolUseMetrics } from "./tool-use-metrics-DEGMKycK.js";
|
|
40
40
|
import { t as analyzeSeries } from "./series-convergence-CjO2QdRW.js";
|
|
41
|
-
import { _ as deterministicSplit, g as BENCHMARK_SPLIT_SEED, t as benchmarks_exports } from "./benchmarks-
|
|
41
|
+
import { _ as deterministicSplit, g as BENCHMARK_SPLIT_SEED, t as benchmarks_exports } from "./benchmarks-v5piCeDl.js";
|
|
42
42
|
import { t as runEvalCampaign } from "./eval-campaign-CvPcvqXC.js";
|
|
43
43
|
import { n as pairedEvalueSequence, t as evaluateInterimReleaseConfidence } from "./sequential-Br0mAPHA.js";
|
|
44
44
|
import { AxGEPA, AxSignature, ax } from "@ax-llm/ax";
|
|
@@ -6792,11 +6792,19 @@ async function evaluateContract(store, contract) {
|
|
|
6792
6792
|
pass: false
|
|
6793
6793
|
};
|
|
6794
6794
|
const samples = [];
|
|
6795
|
+
const measurementBreaches = [];
|
|
6795
6796
|
for (const m of contract.metrics) {
|
|
6796
6797
|
const extract = m.extract ?? defaultExtract$1(m.metric);
|
|
6797
6798
|
const baseline = await extractAll(baselineRuns, extract, store);
|
|
6798
6799
|
const candidate = await extractAll(candidateRuns, extract, store);
|
|
6799
|
-
if (baseline.length
|
|
6800
|
+
if (baseline.length !== baselineRuns.length || candidate.length !== candidateRuns.length) {
|
|
6801
|
+
measurementBreaches.push(`metric "${m.metric}" measured ${candidate.length}/${candidateRuns.length} candidate and ${baseline.length}/${baselineRuns.length} baseline run(s); every supplied run must report the metric`);
|
|
6802
|
+
continue;
|
|
6803
|
+
}
|
|
6804
|
+
if (baseline.length < 2 || candidate.length < 2) {
|
|
6805
|
+
measurementBreaches.push(`metric "${m.metric}" has too few comparable samples: baseline ${baseline.length}, candidate ${candidate.length}; need at least 2 per arm`);
|
|
6806
|
+
continue;
|
|
6807
|
+
}
|
|
6800
6808
|
samples.push({
|
|
6801
6809
|
metric: m.metric,
|
|
6802
6810
|
higherIsBetter: m.higherIsBetter,
|
|
@@ -6811,7 +6819,7 @@ async function evaluateContract(store, contract) {
|
|
|
6811
6819
|
};
|
|
6812
6820
|
let sloReport;
|
|
6813
6821
|
if (contract.slos && contract.slos.length > 0) sloReport = checkSlos(await aggregateRunMetrics(candidateRuns, store), contract.slos);
|
|
6814
|
-
const breaches = [];
|
|
6822
|
+
const breaches = [...measurementBreaches];
|
|
6815
6823
|
for (const metric of baselineReport.metrics) {
|
|
6816
6824
|
const decl = contract.metrics.find((m) => m.metric === metric.metric);
|
|
6817
6825
|
if (!decl) continue;
|
|
@@ -9338,6 +9346,7 @@ function decideReferenceReplayPromotion(baseline, candidate, policy = {}) {
|
|
|
9338
9346
|
const compared = comparisons.filter((item) => requiredSplits.includes(item.split));
|
|
9339
9347
|
const regressions = comparisons.filter((item) => item.f1Delta < -maxRegression);
|
|
9340
9348
|
const aggregateDelta = candidate.aggregate.f1 - baseline.aggregate.f1;
|
|
9349
|
+
const coverageMismatch = compareScenarioCoverage(baseline, candidate);
|
|
9341
9350
|
if (missingRequiredSplits.length > 0) return {
|
|
9342
9351
|
promote: false,
|
|
9343
9352
|
reason: `Required split missing from baseline or candidate: ${missingRequiredSplits.join(", ")}`,
|
|
@@ -9366,6 +9375,16 @@ function decideReferenceReplayPromotion(baseline, candidate, policy = {}) {
|
|
|
9366
9375
|
comparisons,
|
|
9367
9376
|
regressions
|
|
9368
9377
|
};
|
|
9378
|
+
if (coverageMismatch.candidateMissing.length > 0 || coverageMismatch.baselineMissing.length > 0) {
|
|
9379
|
+
const gaps = [coverageMismatch.candidateMissing.length > 0 ? `candidate missing ${formatScenarioIds(coverageMismatch.candidateMissing)}` : null, coverageMismatch.baselineMissing.length > 0 ? `baseline missing ${formatScenarioIds(coverageMismatch.baselineMissing)}` : null].filter((part) => part !== null);
|
|
9380
|
+
return {
|
|
9381
|
+
promote: false,
|
|
9382
|
+
reason: `Scenario coverage differs: baseline ${baseline.scenarios.length}, candidate ${candidate.scenarios.length}; ${gaps.join("; ")}`,
|
|
9383
|
+
aggregateDelta,
|
|
9384
|
+
comparisons,
|
|
9385
|
+
regressions
|
|
9386
|
+
};
|
|
9387
|
+
}
|
|
9369
9388
|
const requiredMeanDelta = mean(compared.map((item) => item.f1Delta));
|
|
9370
9389
|
if (requiredMeanDelta < minF1Delta) return {
|
|
9371
9390
|
promote: false,
|
|
@@ -9382,6 +9401,35 @@ function decideReferenceReplayPromotion(baseline, candidate, policy = {}) {
|
|
|
9382
9401
|
regressions
|
|
9383
9402
|
};
|
|
9384
9403
|
}
|
|
9404
|
+
function compareScenarioCoverage(baseline, candidate) {
|
|
9405
|
+
const baselineCounts = scenarioIdentityCounts(baseline);
|
|
9406
|
+
const candidateCounts = scenarioIdentityCounts(candidate);
|
|
9407
|
+
return {
|
|
9408
|
+
candidateMissing: countDifference(baselineCounts, candidateCounts),
|
|
9409
|
+
baselineMissing: countDifference(candidateCounts, baselineCounts)
|
|
9410
|
+
};
|
|
9411
|
+
}
|
|
9412
|
+
function scenarioIdentityCounts(score) {
|
|
9413
|
+
const counts = /* @__PURE__ */ new Map();
|
|
9414
|
+
for (const scenario of score.scenarios) {
|
|
9415
|
+
const identity = `${scenario.split}:${scenario.scenarioId}`;
|
|
9416
|
+
counts.set(identity, (counts.get(identity) ?? 0) + 1);
|
|
9417
|
+
}
|
|
9418
|
+
return counts;
|
|
9419
|
+
}
|
|
9420
|
+
function countDifference(expected, actual) {
|
|
9421
|
+
const missing = [];
|
|
9422
|
+
for (const [identity, count] of expected) {
|
|
9423
|
+
const difference = count - (actual.get(identity) ?? 0);
|
|
9424
|
+
for (let index = 0; index < difference; index++) missing.push(identity);
|
|
9425
|
+
}
|
|
9426
|
+
return missing.sort();
|
|
9427
|
+
}
|
|
9428
|
+
function formatScenarioIds(ids) {
|
|
9429
|
+
const shown = ids.slice(0, 5);
|
|
9430
|
+
const remaining = ids.length - shown.length;
|
|
9431
|
+
return remaining > 0 ? `${shown.join(", ")}, +${remaining} more` : shown.join(", ");
|
|
9432
|
+
}
|
|
9385
9433
|
function defaultReferenceReplayMatcher(reference, candidate) {
|
|
9386
9434
|
const textScore = tokenJaccard(`${reference.title} ${reference.description ?? ""}`, `${candidate.title} ${candidate.description ?? ""}`);
|
|
9387
9435
|
const severityScore = reference.severity && candidate.severity ? normalize(reference.severity) === normalize(candidate.severity) ? .1 : -.05 : 0;
|
|
@@ -9837,13 +9885,15 @@ function runScore$1(run) {
|
|
|
9837
9885
|
function taskKey(run) {
|
|
9838
9886
|
return run.scenarioId;
|
|
9839
9887
|
}
|
|
9840
|
-
/**
|
|
9888
|
+
/** Tasks one arm was dealt and the mean score for each task it answered. */
|
|
9841
9889
|
function perTaskMeanScore(runs) {
|
|
9890
|
+
const dealt = /* @__PURE__ */ new Set();
|
|
9842
9891
|
const acc = /* @__PURE__ */ new Map();
|
|
9843
9892
|
for (const run of runs) {
|
|
9893
|
+
const key = taskKey(run);
|
|
9894
|
+
dealt.add(key);
|
|
9844
9895
|
const s = runScore$1(run);
|
|
9845
9896
|
if (s === void 0) continue;
|
|
9846
|
-
const key = taskKey(run);
|
|
9847
9897
|
const cur = acc.get(key) ?? {
|
|
9848
9898
|
sum: 0,
|
|
9849
9899
|
n: 0
|
|
@@ -9852,7 +9902,10 @@ function perTaskMeanScore(runs) {
|
|
|
9852
9902
|
cur.n += 1;
|
|
9853
9903
|
acc.set(key, cur);
|
|
9854
9904
|
}
|
|
9855
|
-
return
|
|
9905
|
+
return {
|
|
9906
|
+
dealt,
|
|
9907
|
+
answered: new Map([...acc].map(([key, value]) => [key, value.sum / value.n]))
|
|
9908
|
+
};
|
|
9856
9909
|
}
|
|
9857
9910
|
/** Compressed-model bits — the model's description length L_model. gzip is a
|
|
9858
9911
|
* deterministic, dependency-free stand-in for Kolmogorov complexity; it
|
|
@@ -9866,7 +9919,7 @@ function dataDescriptionBits(scoreByTask, keys, scoreFloor) {
|
|
|
9866
9919
|
let bits = 0;
|
|
9867
9920
|
for (const key of keys) {
|
|
9868
9921
|
const s = scoreByTask.get(key);
|
|
9869
|
-
if (s === void 0)
|
|
9922
|
+
if (s === void 0) throw new ValidationError(`dataDescriptionBits: no score for task '${key}'`);
|
|
9870
9923
|
bits += -Math.log2(Math.max(s, scoreFloor));
|
|
9871
9924
|
}
|
|
9872
9925
|
return bits;
|
|
@@ -9892,9 +9945,14 @@ var DescriptionLengthGate = class {
|
|
|
9892
9945
|
* if λ·L_model + L_data is strictly lower by at least `marginBits`. */
|
|
9893
9946
|
evaluate(candidate, baseline) {
|
|
9894
9947
|
const candidateId = inferCandidateId$1(candidate.runs, this.baselineKey);
|
|
9895
|
-
const
|
|
9896
|
-
const
|
|
9897
|
-
const
|
|
9948
|
+
const candidateTasks = perTaskMeanScore(candidate.runs);
|
|
9949
|
+
const baselineTasks = perTaskMeanScore(baseline.runs);
|
|
9950
|
+
const candScores = candidateTasks.answered;
|
|
9951
|
+
const baseScores = baselineTasks.answered;
|
|
9952
|
+
const dealt = [.../* @__PURE__ */ new Set([...candidateTasks.dealt, ...baselineTasks.dealt])].sort();
|
|
9953
|
+
const shared = dealt.filter((key) => candScores.has(key) && baseScores.has(key));
|
|
9954
|
+
const candidateMissing = dealt.filter((key) => !candScores.has(key));
|
|
9955
|
+
const baselineMissing = dealt.filter((key) => !baseScores.has(key));
|
|
9898
9956
|
const modelBits = {
|
|
9899
9957
|
candidate: this.lambda * modelDescriptionBits(candidate.content),
|
|
9900
9958
|
baseline: this.lambda * modelDescriptionBits(baseline.content)
|
|
@@ -9928,6 +9986,15 @@ var DescriptionLengthGate = class {
|
|
|
9928
9986
|
reason: `few_tasks: ${shared.length} shared task(s) < min ${this.minTasks}`,
|
|
9929
9987
|
rejectionCode: "few_tasks"
|
|
9930
9988
|
};
|
|
9989
|
+
if (candidateMissing.length > 0 || baselineMissing.length > 0) {
|
|
9990
|
+
const missing = [candidateMissing.length > 0 ? `candidate answered ${candScores.size}/${dealt.length}, ${candidateMissing.length} missing: ${formatTaskIds(candidateMissing)}` : null, baselineMissing.length > 0 ? `baseline answered ${baseScores.size}/${dealt.length}, ${baselineMissing.length} missing: ${formatTaskIds(baselineMissing)}` : null].filter((part) => part !== null);
|
|
9991
|
+
return {
|
|
9992
|
+
...base,
|
|
9993
|
+
promote: false,
|
|
9994
|
+
reason: `missing_task_scores: ${missing.join("; ")}. The decision cannot use only the ${shared.length} task(s) both arms answered`,
|
|
9995
|
+
rejectionCode: "missing_task_scores"
|
|
9996
|
+
};
|
|
9997
|
+
}
|
|
9931
9998
|
if (!(evidence.deltaBits < -this.marginBits)) {
|
|
9932
9999
|
const code = evidence.dataBitsDelta < 0 ? "model_bloat" : "no_total_gain";
|
|
9933
10000
|
const why = code === "model_bloat" ? `model grew ${fmt(evidence.modelBitsDelta)} bits, outpacing a ${fmt(-evidence.dataBitsDelta)}-bit data gain` : `outcomes did not improve (data Δ=${fmt(evidence.dataBitsDelta)} bits)`;
|
|
@@ -9946,6 +10013,11 @@ var DescriptionLengthGate = class {
|
|
|
9946
10013
|
};
|
|
9947
10014
|
}
|
|
9948
10015
|
};
|
|
10016
|
+
function formatTaskIds(ids) {
|
|
10017
|
+
const shown = ids.slice(0, 5);
|
|
10018
|
+
const remaining = ids.length - shown.length;
|
|
10019
|
+
return remaining > 0 ? `${shown.join(", ")}, +${remaining} more` : shown.join(", ");
|
|
10020
|
+
}
|
|
9949
10021
|
function inferCandidateId$1(runs, baselineKey) {
|
|
9950
10022
|
for (const run of runs) if (run.candidateId && run.candidateId !== baselineKey) return run.candidateId;
|
|
9951
10023
|
return runs[0]?.candidateId ?? "(unknown candidate)";
|
|
@@ -10089,8 +10161,11 @@ function precision(goldens, candidates, options = {}) {
|
|
|
10089
10161
|
* The optimizer's best-guess is one thing; what we should actually
|
|
10090
10162
|
* ship is another. The gate is the line between them.
|
|
10091
10163
|
*
|
|
10092
|
-
* A candidate is promoted iff ALL
|
|
10164
|
+
* A candidate is promoted iff ALL FOUR pass:
|
|
10093
10165
|
*
|
|
10166
|
+
* 0. **Coverage**: on BOTH splits, the fraction of DEALT work items that
|
|
10167
|
+
* carry a real score on BOTH arms is at least `minCoverage`
|
|
10168
|
+
* (default 1 — all of them). See "Missing scores are missing data".
|
|
10094
10169
|
* 1. **Productive runs**: the candidate has at least
|
|
10095
10170
|
* `minProductiveRuns` paired observations on items where BOTH
|
|
10096
10171
|
* candidate and baseline produced a real (non-silent) score.
|
|
@@ -10103,6 +10178,36 @@ function precision(goldens, candidates, options = {}) {
|
|
|
10103
10178
|
* "Better on search, worse on holdout" is the canonical
|
|
10104
10179
|
* overfit pattern; this catches it.
|
|
10105
10180
|
*
|
|
10181
|
+
* ## Missing scores are missing data
|
|
10182
|
+
*
|
|
10183
|
+
* Gates 1-3 all read numbers off the items where BOTH arms produced a finite
|
|
10184
|
+
* score. Without gate 0 that set is a SILENT FILTER: an item the candidate
|
|
10185
|
+
* crashed on, timed out on, or never wrote a row for simply left the
|
|
10186
|
+
* comparison, and the verdict was computed over whatever survived. Measured on
|
|
10187
|
+
* the pre-0.134 gate: 26 held-out items, a candidate that produced no score at
|
|
10188
|
+
* all on 20 of them, promoted at every threshold from −0.05 to +0.30 on the 6
|
|
10189
|
+
* it answered (`unpairedBaselineRuns: 20` sat in the evidence, read by
|
|
10190
|
+
* nothing); the control — the same 20 failures scored as the 0 they earned —
|
|
10191
|
+
* was correctly refused at a mean paired delta of −0.3808. An agent that fails
|
|
10192
|
+
* 77% of its tasks was promoted, and every guarantee gates 1-3 provide sat
|
|
10193
|
+
* behind that filter.
|
|
10194
|
+
*
|
|
10195
|
+
* The gate does NOT impute a value for those items. It cannot: it does not know
|
|
10196
|
+
* the failure value of the caller's metric (0 on [0,1]? 0 on 0-100? a
|
|
10197
|
+
* lower-is-better latency?), and guessing one is exactly the silent assumption
|
|
10198
|
+
* this gate exists to remove. A caller who DOES know it writes it onto the
|
|
10199
|
+
* record before calling — that is the caller's decision to make and to be
|
|
10200
|
+
* accountable for, and it then travels in the run corpus where an auditor can
|
|
10201
|
+
* see it. What the gate does instead is REFUSE, and report the full coverage
|
|
10202
|
+
* accounting (`GateEvidence.holdoutCoverage` / `searchCoverage`) so the caller
|
|
10203
|
+
* can see exactly which items went dark and on which arm.
|
|
10204
|
+
*
|
|
10205
|
+
* The denominator is MEASURED, never assumed: it is the set of work items the
|
|
10206
|
+
* comparison was DEALT, which is what `pairRunRecords` reports when it is given
|
|
10207
|
+
* every row of a split rather than only the scored ones. There is no list of
|
|
10208
|
+
* "expected" scenarios to keep in sync and nothing to trust — an item counts as
|
|
10209
|
+
* dealt because a row for it exists on at least one arm.
|
|
10210
|
+
*
|
|
10106
10211
|
* The decision carries a machine-readable `rejectionCode` plus an
|
|
10107
10212
|
* `evidence` block with every number the gate looked at, so the
|
|
10108
10213
|
* downstream researcher / paper / dashboard can re-derive the
|
|
@@ -10128,8 +10233,11 @@ var HeldOutGate = class {
|
|
|
10128
10233
|
resamples;
|
|
10129
10234
|
seed;
|
|
10130
10235
|
costPerTaskCeiling;
|
|
10236
|
+
minCoverage;
|
|
10131
10237
|
constructor(config) {
|
|
10132
10238
|
if (!config.baselineKey) throw new Error("HeldOutGate: baselineKey is required");
|
|
10239
|
+
if (config.minCoverage !== void 0 && !(Number.isFinite(config.minCoverage) && config.minCoverage >= 0 && config.minCoverage <= 1)) throw new Error(`HeldOutGate: minCoverage must be a finite fraction in [0, 1], got ${config.minCoverage}`);
|
|
10240
|
+
this.minCoverage = config.minCoverage ?? 1;
|
|
10133
10241
|
this.confidence = config.confidence ?? .95;
|
|
10134
10242
|
this.minProductiveRuns = Math.max(config.minProductiveRuns ?? 3, minimumPairsForPairedDeltaTest(this.confidence));
|
|
10135
10243
|
this.pairedDeltaThreshold = config.pairedDeltaThreshold ?? 0;
|
|
@@ -10156,6 +10264,8 @@ var HeldOutGate = class {
|
|
|
10156
10264
|
const baselineHoldout = scoredRuns(honestBaseline, "holdout");
|
|
10157
10265
|
const searchPairing = pairRunRecords(baselineSearch, candidateSearch);
|
|
10158
10266
|
const holdoutPairing = pairRunRecords(baselineHoldout, candidateHoldout);
|
|
10267
|
+
const searchCoverage = splitCoverage(honestBaseline, honestCandidate, "search");
|
|
10268
|
+
const holdoutCoverage = splitCoverage(honestBaseline, honestCandidate, "holdout");
|
|
10159
10269
|
const splitScoreOf = (run, split) => observedSplitScore(run, split);
|
|
10160
10270
|
const beforeSearch = searchPairing.pairs.map((pair) => splitScoreOf(pair.baseline, "search"));
|
|
10161
10271
|
const afterSearch = searchPairing.pairs.map((pair) => splitScoreOf(pair.treatment, "search"));
|
|
@@ -10168,8 +10278,8 @@ var HeldOutGate = class {
|
|
|
10168
10278
|
const baselineHoldoutMean = meanOrNull(beforeHoldout);
|
|
10169
10279
|
const overfitGap = diffOrNull(candidateSearchMean, candidateHoldoutMean);
|
|
10170
10280
|
const baselineOverfitGap = diffOrNull(baselineSearchMean, baselineHoldoutMean);
|
|
10171
|
-
const medianCandidateCost = completeCostMedian(
|
|
10172
|
-
const medianBaselineCost = completeCostMedian(baseline);
|
|
10281
|
+
const medianCandidateCost = completeCostMedian([...searchPairing.pairs.map((pair) => pair.treatment), ...holdoutPairing.pairs.map((pair) => pair.treatment)]);
|
|
10282
|
+
const medianBaselineCost = completeCostMedian([...searchPairing.pairs.map((pair) => pair.baseline), ...holdoutPairing.pairs.map((pair) => pair.baseline)]);
|
|
10173
10283
|
const commonEvidence = {
|
|
10174
10284
|
productiveRuns,
|
|
10175
10285
|
unpairedCandidateRuns: holdoutPairing.unpairedTreatment.length,
|
|
@@ -10180,7 +10290,9 @@ var HeldOutGate = class {
|
|
|
10180
10290
|
baselineOverfitGap,
|
|
10181
10291
|
medianCandidateCost,
|
|
10182
10292
|
medianBaselineCost,
|
|
10183
|
-
realnessGatedRuns
|
|
10293
|
+
realnessGatedRuns,
|
|
10294
|
+
searchCoverage,
|
|
10295
|
+
holdoutCoverage
|
|
10184
10296
|
};
|
|
10185
10297
|
const missingSplitScores = [
|
|
10186
10298
|
candidateSearch.length === 0 ? "candidate search" : null,
|
|
@@ -10201,6 +10313,20 @@ var HeldOutGate = class {
|
|
|
10201
10313
|
reason: `missing_split_scores: ${missingSplitScores.join(", ")} score evidence is absent`,
|
|
10202
10314
|
rejectionCode: "missing_split_scores"
|
|
10203
10315
|
};
|
|
10316
|
+
const shortfalls = [["holdout", holdoutCoverage], ["search", searchCoverage]].filter(([, cov]) => cov.coverage < this.minCoverage);
|
|
10317
|
+
if (shortfalls.length > 0) return {
|
|
10318
|
+
promote: false,
|
|
10319
|
+
candidateId,
|
|
10320
|
+
baselineId,
|
|
10321
|
+
evidence: {
|
|
10322
|
+
...commonEvidence,
|
|
10323
|
+
medianPairedDelta: productiveRuns > 0 ? medianDelta(beforeHoldout, afterHoldout) : null,
|
|
10324
|
+
pairedCI: null,
|
|
10325
|
+
pairedPValue: null
|
|
10326
|
+
},
|
|
10327
|
+
reason: `incomplete_coverage: ${shortfalls.map(([split, cov]) => describeCoverage(split, cov)).join("; ")} — required ${fmt(this.minCoverage)} of every dealt item scored on both arms. A missing score is missing data, not an absent item: the gate will not decide on the subset that answered`,
|
|
10328
|
+
rejectionCode: "incomplete_coverage"
|
|
10329
|
+
};
|
|
10204
10330
|
if (productiveRuns < this.minProductiveRuns) return {
|
|
10205
10331
|
promote: false,
|
|
10206
10332
|
candidateId,
|
|
@@ -10304,6 +10430,50 @@ function scoredRuns(runs, split) {
|
|
|
10304
10430
|
return typeof v === "number" && Number.isFinite(v);
|
|
10305
10431
|
});
|
|
10306
10432
|
}
|
|
10433
|
+
/** Every row tagged to one split, scored or not — the work the split was
|
|
10434
|
+
* DEALT, as opposed to the work it managed to measure. */
|
|
10435
|
+
function dealtRuns(runs, split) {
|
|
10436
|
+
return runs.filter((run) => run.splitTag === split);
|
|
10437
|
+
}
|
|
10438
|
+
function hasFiniteScore(run, split) {
|
|
10439
|
+
const v = observedSplitScore(run, split);
|
|
10440
|
+
return typeof v === "number" && Number.isFinite(v);
|
|
10441
|
+
}
|
|
10442
|
+
/**
|
|
10443
|
+
* Partition one split's dealt work into answered / unscored / one-armed.
|
|
10444
|
+
*
|
|
10445
|
+
* Pairs the UNFILTERED rows with the same `pairRunRecords` the decision uses,
|
|
10446
|
+
* so the denominator comes out of the same identity rules as the numerator and
|
|
10447
|
+
* cannot drift from it. Deliberately a SECOND pairing rather than a widening of
|
|
10448
|
+
* the decision's own: the decision keeps pairing exactly the rows it always
|
|
10449
|
+
* paired, so this check can only ever add a refusal, never move a verdict that
|
|
10450
|
+
* was already being made on a complete set.
|
|
10451
|
+
*/
|
|
10452
|
+
function splitCoverage(baseline, candidate, split) {
|
|
10453
|
+
const pairing = pairRunRecords(dealtRuns(baseline, split), dealtRuns(candidate, split));
|
|
10454
|
+
let answered = 0;
|
|
10455
|
+
for (const pair of pairing.pairs) if (hasFiniteScore(pair.baseline, split) && hasFiniteScore(pair.treatment, split)) answered += 1;
|
|
10456
|
+
const unscoredPairs = pairing.pairs.length - answered;
|
|
10457
|
+
const candidateOnly = pairing.unpairedTreatment.length;
|
|
10458
|
+
const baselineOnly = pairing.unpairedBaseline.length;
|
|
10459
|
+
const dealt = pairing.pairs.length + candidateOnly + baselineOnly;
|
|
10460
|
+
return {
|
|
10461
|
+
dealt,
|
|
10462
|
+
answered,
|
|
10463
|
+
unscoredPairs,
|
|
10464
|
+
candidateOnly,
|
|
10465
|
+
baselineOnly,
|
|
10466
|
+
coverage: dealt === 0 ? 0 : answered / dealt
|
|
10467
|
+
};
|
|
10468
|
+
}
|
|
10469
|
+
function describeCoverage(split, cov) {
|
|
10470
|
+
const dark = [
|
|
10471
|
+
cov.unscoredPairs > 0 ? `${cov.unscoredPairs} ran on both arms but produced no score` : null,
|
|
10472
|
+
cov.baselineOnly > 0 ? `${cov.baselineOnly} only the baseline produced a row for` : null,
|
|
10473
|
+
cov.candidateOnly > 0 ? `${cov.candidateOnly} only the candidate produced a row for` : null
|
|
10474
|
+
].filter((part) => part !== null);
|
|
10475
|
+
return `${split} scored ${cov.answered} of ${cov.dealt} dealt items (coverage ${fmt(cov.coverage)})` + (dark.length > 0 ? ` — ${dark.join(", ")}` : "");
|
|
10476
|
+
}
|
|
10307
10477
|
function meanOrNull(xs) {
|
|
10308
10478
|
if (xs.length === 0) return null;
|
|
10309
10479
|
return xs.reduce((s, x) => s + x, 0) / xs.length;
|
|
@@ -12123,6 +12293,6 @@ function assertProductBenchmarkRun(runDir) {
|
|
|
12123
12293
|
return report;
|
|
12124
12294
|
}
|
|
12125
12295
|
//#endregion
|
|
12126
|
-
export { AGENT_PROFILE_KINDS, ATIF_SCHEMA_VERSION, ATTESTATION_ALGORITHM, AgentDriver, AgentEvalError, AgentProfileCellValidationError, AnalystRegistry, AxGepaSteeringOptimizer, BENCHMARK_SPLIT_SEED, BOOTSTRAP_GATE_MIN_N, BackendIntegrityError, BenchmarkRunner, BudgetBreachError, BudgetGuard, CODING_HARNESSES, CallbackResearcher, CaptureIntegrityError, ConfigError, ConvergenceTracker, CostAccountingIncompleteError, CostCallConflictError, CostCeilingReachedError, CostLedger, CostLedgerPersistenceError, CostReceiptCaptureError, CostReservationExceededError, CostTracker, CrossFamilyError, DEFAULT_AGENT_SLOS, DEFAULT_COMPLEXITY_WEIGHTS, DEFAULT_RULES as DEFAULT_FAILURE_RULES, DEFAULT_FINDERS, DEFAULT_HARNESS_OBJECTIVES, DEFAULT_MUTATION_PRIMITIVES, DEFAULT_MUTATORS, DEFAULT_PERMUTATIONS, DEFAULT_PR_REVIEW_SCORE_WEIGHTS, DEFAULT_REDACTION_RULES, DEFAULT_RED_TEAM_CORPUS, DEFAULT_RUN_SCORE_WEIGHTS, DEFAULT_SEVERITY_WEIGHTS, DEFAULT_TRACE_ANALYST_BUDGETS, DEFAULT_TRACE_ANALYST_KINDS, Dataset, DescriptionLengthGate, DockerSandboxDriver, DualAgentBench, ERROR_COUNT_PATTERNS, EvalTraceStore, ExperimentTracker, FAILURE_CLASSES, FAILURE_MODE_KIND_SPEC, FileSystemFeedbackTrajectoryStore, FileSystemRawProviderSink, FileSystemTraceStore, FindingsStore, HARBOR_IMPORT_GAP, HARNESS_NATIVE_MODEL, HeldOutGate, HoldoutAuditor, HoldoutLockedError, IMPROVEMENT_KIND_SPEC, INPUT_VALUE, INTENT_MATCH_JUDGE_VERSION, InMemoryFeedbackTrajectoryStore, InMemoryRawProviderSink, InMemoryTraceStore, InMemoryWorkspaceInspector, JudgeError, JudgeParseError, JudgeRunner, KNOWLEDGE_GAP_KIND_SPEC, KNOWLEDGE_POISONING_KIND_SPEC, LLM_CACHED_TOKENS, LLM_CACHED_TOKEN_ATTR_KEYS, LLM_CACHE_WRITE_TOKENS, LLM_CACHE_WRITE_TOKEN_ATTR_KEYS, LLM_CONTEXT_TOKENS, LLM_COST_ATTR_KEYS, LLM_COST_USD, LLM_INPUT_TOKENS, LLM_INPUT_TOKEN_ATTR_KEYS, LLM_MODEL_ATTR_KEYS, LLM_MODEL_NAME, LLM_OUTPUT_TOKENS, LLM_OUTPUT_TOKEN_ATTR_KEYS, LLM_REASONING_TOKENS, LLM_REASONING_TOKEN_ATTR_KEYS, LlmCallError, LlmClient, LlmResponseError, LlmRouteAssertionError, LockedJsonlAppender, MANN_WHITNEY_EXACT_MAX_STATES, MANN_WHITNEY_EXACT_MAX_WORK, MODEL_PRICING, MetricsCollector, ModelsUnreachableError, MultiLayerVerifier, Mutex, NoopRawProviderSink, NoopResearcher, NotFoundError, OPENINFERENCE_SPAN_KIND, OTEL_AGENT_EVAL_SCOPE, OUTPUT_VALUE, OtlpFileTraceStore, PairwiseSteeringOptimizer, ProductClient, PromptRegistry, REDACTION_VERSION, REFERENCE_EQUIVALENCE_INPUT_LIMITS, REFERENCE_EQUIVALENCE_JUDGE_VERSION, RESEARCH_REPORT_HARD_PAIR_FLOOR, ROLLOUT_SCHEMA, RUN_COST_ATTR_KEYS, ReplayCache, ReplayCacheMissError, ReplayError, RunCritic, RunIntegrityError, RunRecordValidationError, SEMANTIC_CONCEPT_JUDGE_VERSION, SKILL_USAGE_ANALYST, SPAN_KIND_ATTR_KEYS, SUPERVISOR_RUN_SCHEMA, SandboxHarness, ScenarioRegistry, SeatUnsetError, SingleBackendError, SkillUsageAnalyst, SpanNotFoundError, SubprocessSandboxDriver, TOOL_ARGS_CAPTURED, TOOL_LATENCY_MS, TOOL_NAME, TOOL_NAME_ATTR_KEYS, TRACE_ANALYST_ACTOR_DESCRIPTION, TRACE_ANALYST_ACTOR_DESCRIPTION_VERSION, TRACE_ANALYST_TRUNCATION_MARKER_PREFIX, TRACE_SCHEMA_VERSION, TokenCounter, TraceContractBuilder, TraceEmitter, TraceFileMissingError, TraceNotFoundError, UI_FINDING_SEVERITIES, UI_LENSES, UNIVERSAL_FINDERS, ValidationError, VerificationError, WILCOXON_EXACT_MAX_N, acquisitionPlansForKnowledgeGaps, agentProfileCellHashMaterial, agentProfileCellKey, agentProfileHash, agentProfileId, agentProfileModelId, agentVisibleFields, aggregateJudgeVerdicts, aggregateLlm, aggregatePrReviewScore, aggregateRunScore, allCriticalPassed, analyzeAntiSlop, analyzeSeries, analyzeSupervisorRun, analyzeSupervisorRunSources, analyzeTraces, appendScorecard, applyLlmSpanOtlpAttributes, applyToolSpanOtlpAttributes, argHash, asNumber, asString, assertCapabilityHeadroom, assertCrossFamily, assertLlmRoute, assertMinted, assertMintedLines, assertModelsServed, assertNoHiddenLeak, assertProductBenchmarkRun, assertRealAgentReceipts, assertRealBackend, assertReleaseConfidence, assertRolloutLine, assertRunAgentProfileCell, assertRunCaptured, assertSingleBackend, assignFeedbackSplit, assignHeldOutTag, attachCostToReport, attest, attributeCounterfactuals, backoffMs, deterministicSplit as benchmarkDeterministicSplit, benchmarks_exports as benchmarks, benjaminiHochberg, bisect, blendHeldout, blockingKnowledgeEval, bonferroni, bootstrapCi, buildAgentInterfaceProfileCell, buildAgentProfileCell, buildDefaultAnalystRegistry, buildDriverSystemPrompt, buildProductBenchmarkManifest, buildReflectionPrompt, buildReviewerPrompt, buildTraceAnalystTools, buildTraceInsightContext, buildTraceInsightPrompt, buildTrajectory, buildWorkerDriverSystemPrompt, byteLengthRange, cachedJudge, calibrateJudge, calibrateJudgeContinuous, callLlm, callLlmJson, canaryLeakView, canonicalJson, canonicalize, capabilityHeadroom, captureFetchToRawSink, causalAttribution, checkBehavioralCanary, checkCanaries, checkSlos, checkTraceContracts, clamp01, classifyFailure, classifyOtlpSpanRole, classifyTreatment, claudeCodeSupervisorRunReader, cliffsDelta, clusteredPairedBinary, cohensD, collectionPreserved, commentsForSource, commitBisect, comparePairedArms, compareReferenceReplay, compareToBaseline, compilerJudge, completionVerdict, composeParsers, composeValidators, computeExperimentStats, computeFindingId, computeToolUseMetrics, computeTraceMetrics, confidenceInterval, containsAll, contentHash, contextInputTokens, continuousAgreement, contractJudge, controlFailureClassFromVerification, controlRunToFeedbackTrajectory, controlRunToRunRecord, convertTraceStoresToOtlp, corpusInterRaterAgreement, corpusInterRaterAgreementFromJudgeScores, costForTokenPricing, costForUsage, costReceiptFromLlm, costReceiptFromLlmError, costReport, createAnalystAi, createAntiSlopJudge, createChatClient, createDefaultReviewer, createFeedbackTrajectory, createIntentMatchJudge, createLlmCorrectnessChecker, createLlmReviewer, createOtelExporter, createOtelTracingStore, createReferenceEquivalenceJudge, createReplayFetch, createSandboxPool, createSemanticConceptJudge, createTokenRecallChecker, createTraceAnalystKind, crossTraceDiff, crowdingDistance, dataDescriptionBits, decideNextUserTurn, decideReferenceReplayPromotion, decideReferenceReplayRunPromotion, defaultBlendWeights, defaultIsMaterial, defaultProviderRedactor, defaultReferenceReplayMatcher, defaultTraceInsightPanel, deployGateLayer, describeTraceInsightScope, diffFindings, diffScorecard, discoverPersonas, distillPlaybook, domainEvidencePattern, dominates, eProcess, ensembleJudge, errorStreakDetector, estimateCost, estimateTokens, evaluateActionPolicy, evaluateContract, evaluateHypothesis, evaluateInterimReleaseConfidence, evaluateOracles, evaluateReleaseConfidence, evaluateTraceContract, executeScenario, expandProfileAxes, expectAgent, exportProductBenchmark, exportProductBenchmarkRuns, exportRewardModel, exportRunAsOtlp, extractAssetUrls, extractErrorCount, extractOtlpAttributes, extractProducedState, extractUsage, extractUsageFromResponse, extractUsageFromSse, feedbackTrajectoriesToDatasetScenarios, feedbackTrajectoriesToOptimizerRows, feedbackTrajectoryToDatasetScenario, feedbackTrajectoryToOptimizerRow, fileContains, fileExists, fileExperimentStore, fileVerdictCache, findAutoMatchNoExpectation, findConstructorCwdDropped, findFallbackToPass, findLiteralTruePass, findProductBenchmarkArtifacts, findSkipCountsAsPass, firstNumberAttr, firstStringAttr, flattenOtlpExportToNdjson, flowLayer, fnv1a32, formatBenchmarkReport, formatDriverReport, formatFindings, formatScorecardDiff, fromHarborTrajectory, gainHistogram, gateTreatmentApplied, gateTreatmentFromMetrics, gateTreatmentFromSpans, gateTreatmentFromToolSpans, ghCliClient, gitProvenanceReader, precision as goldenPrecision, gradeOnHidden, gradeSemanticStatus, groupBy, groupRunsByAgentProfileCell, harnessAxisOf, hasCapturedToolArgs, hashContent, hashJson, hashScenarios, hashToUnit, hiddenGrade, holm, htmlContainsElement, httpGithubClient, improvementVerdict, inMemoryExperimentStore, inMemoryReferenceReplayStore, inMemoryReviewStore, inMemoryRunRecordBackend, inMemoryVerdictCache, inferDomainKeywords, inferOtlpKind, interRaterReliability, interpretCliffs, iqr, isHiddenDestination, isJudgeSpan, isLlmSpan, isModelPriced, isOtelConfigured, isOtlpModelCall, isRealnessGated, isRetrievalSpan, isRolloutLine, isRunRecord, isSandboxSpan, isToolSpan, isTrainableSplit, isTransientLlmError, isUnavailable, iterateRawCalls, jestTestParser, jsonHasKeys, jsonShape, jsonlReferenceReplayStore, jsonlReviewStore, jsonlRunRecordBackend, judgeFamily, judgeReplayGate, judgeSpans, keyPreserved, knowledgeReadinessTracePayload, leaderboard, linterJudge, llmJudge, llmSpanFromProvider, llmSpans, loadScorecard, loadScorerFromGrader, localCommandRunner, lowercaseMutator, makeEvalTools, makeFinding, mannWhitneyU, mapConcurrent, matchGoldens, matchSpan, maximumChargeForLlmRequest, mcnemar, mcnemarPower, mcnemarRequiredN, mergeLayerResults, mergeSteeringBundle, minimumPairsForPairedDeltaTest, mintRolloutRows, modelDescriptionBits, modelHasSnapshot, modelPriceKey, mulberry32, multiToolchainLayer, noProgressDetector, normalCdf, normalizeScores, notBlocked, objectiveEval, observeAll, observedScore, observedSplitScore, otelRunCompleteHook, otlpRowsToRunRecords, otlpRowsToTraceRunRecords, otlpToRunRecords, otlpToTraceRunRecords, pairArms, pairRunRecords, pairedBootstrap, pairedCohensDz, pairedDeltaTest, pairedEvalueSequence, pairedMde, pairedRiskDifference, pairedSignTest, pairedTTest, paraphraseRobustness, paraphraseRobustnessScenarios, paretoChart, paretoFrontier, paretoFrontierWithCrowding, parseCorrectnessResponse, parseFeedbackTrajectoriesJsonl, parseReflectionResponse, parseRunRecordSafe, parseRuntimeTrajectoryHookEvent, partialCredit, partitionHeldOut, passAtK, passOrthogonality, pearsonR, pixelDeltaRatio, planTraceInsightQuestions, politenessPrefixMutator, positionalBias, preflightModels, printDriverSummary, probeLlm, productBenchmarkIntegrityFailures, productBenchmarkMutableSurfaces, productBenchmarkRepoIdentity, productBenchmarkSplits, profile_exports as profile, projectOtlpFlatLine, projectRuntimeTrajectoryEvidence, promptBisect, proposeSynthesisTargets, providerFromBaseUrl, pytestTestParser, ranks, readClaudeCodeSupervisorRun, readOtlpStatus, readProductBenchmarkManifest, readProductBenchmarkRecords, recordRuns, recordRunsToScorecard, redTeamDataset, redTeamReport, redactString, redactValue, referenceReplayRunsToSteeringRows, referenceReplayScenarioToRunScore, regexMatch, regexMatches, relabelImportedSplit, renderMarkdownReport, renderPlaybookMarkdown, renderPreferenceMemoryMarkdown, renderPriorFindings, renderReleaseReport, renderSteeringText, renderSupervisorRunHeadline, renderSupervisorRunMarkdown, renderUpstreamFindings, repeatedActionDetector, replayFeedbackTrajectories, replayFeedbackTrajectory, replayScorerOverCorpus, replayTraceThroughJudge, requireAgentProfileCell, requiredPairedSampleSize, requiredSampleSize, researchReport, resolveModelPricing, resolveSeat, rollupSupervisorRuns, roundTripRunRecord, routeFields, rowCount, rowWhere, runAgentControlLoop, runAssertions, runBehavioralCanaries, runCanaries, runCounterfactual, runE2EWorkflow, runEvalCampaign, runExpectations, runFailureClass, runHarnessExperiment, runIntentMatchJudge, runJudgeFleet, runKeywordCoverageJudge, runKeywordCoverageJudgeUrl, runLiveProof, runProposeReview, runProposeReviewAsControlLoop, runRecordToProductBenchmarkRecord, runReferenceEquivalenceJudge, runReferenceReplay, runScore, runSelfPlay, runSemanticConceptJudge, runTaskScore, runTestGradedScenario, runsForScenario, scalarScore, scanForMuffledGates, scoreContinuity, scoreFromEvals, scoreKnowledgeReadiness, scoreOrigin, scorePrReviewComments, scorePrReviewSource, scoreRedTeamOutput, scoreReferenceReplay, scoreTraceInsightReadiness, seatPresets, securityJudge, selectHarnessVariant, selfPreference, sentenceReorderMutator, serializeFeedbackTrajectoriesJsonl, showMeasured, signManifest, spearmanR, statusAdvanced, stopOnNoProgress, stopOnRepeatedAction, stringField, stripFencedJson, studentTCdf, subjectiveEval, summarizeAgentReceiptIntegrity, summarizeBackendIntegrity, summarizeHarnessResults, summarizePrReviewBenchmark, summarizePreferenceMemory, summaryTable, supervisorRunRolloutLines, testJudge, textInSnapshot, throwIfRunIncomplete, toAgentProfileJson, toHarborTrajectories, toHarborTrajectory, toJsonl, toLangfuseEnvelope, toOpenAiTool, toPrometheusText, toRewardRows, toSftRows, tokenizeDomainWords, toolNamesForRun, toolSpans, traceAnalystFunctionGroup, traceAnalystOnRunComplete, traceContract, traceJudge, traceJudgeEnsemble, traceSpanKindToOpenInferenceKind, tracedAnalyzeTraces, trainingReward, trainingScore, typoMutator, urlContains, userQuestionsForKnowledgeGaps, validateAgentProfileCell, validateProductBenchmarkManifest, validateProductBenchmarkRecord, validateProductBenchmarkRun, validateRolloutLine, validateRunRecord, verbosityBias, verifyAgentProfileCell, verifyAttestation, verifyCompletion, verifyManifest, visualDiff, viteDeployRunner, vitestTestParser, weightedComposite, weightedMean, weightedRecall, welchsTTest, whitespaceCollapseMutator, wilcoxonSignedRank, wilson, withAssignedFeedbackSplit, withHeldoutBlend, withJudgeRetry, withOtelPipeline, wranglerDeployRunner, writeSupervisorRunReport };
|
|
12296
|
+
export { AGENT_PROFILE_KINDS, ATIF_SCHEMA_VERSION, ATTESTATION_ALGORITHM, AgentDriver, AgentEvalError, AgentProfileCellValidationError, AnalystRegistry, AxGepaSteeringOptimizer, BENCHMARK_SPLIT_SEED, BOOTSTRAP_GATE_MIN_N, BackendIntegrityError, BenchmarkRunner, BudgetBreachError, BudgetGuard, CODING_HARNESSES, CallbackResearcher, CaptureIntegrityError, ConfigError, ConvergenceTracker, CostAccountingIncompleteError, CostCallConflictError, CostCeilingReachedError, CostLedger, CostLedgerPersistenceError, CostReceiptCaptureError, CostReservationExceededError, CostTracker, CrossFamilyError, DEFAULT_AGENT_SLOS, DEFAULT_COMPLEXITY_WEIGHTS, DEFAULT_RULES as DEFAULT_FAILURE_RULES, DEFAULT_FINDERS, DEFAULT_HARNESS_OBJECTIVES, DEFAULT_MUTATION_PRIMITIVES, DEFAULT_MUTATORS, DEFAULT_PERMUTATIONS, DEFAULT_PR_REVIEW_SCORE_WEIGHTS, DEFAULT_REDACTION_RULES, DEFAULT_RED_TEAM_CORPUS, DEFAULT_RUN_SCORE_WEIGHTS, DEFAULT_SEVERITY_WEIGHTS, DEFAULT_TRACE_ANALYST_BUDGETS, DEFAULT_TRACE_ANALYST_KINDS, Dataset, DescriptionLengthGate, DockerSandboxDriver, DualAgentBench, ERROR_COUNT_PATTERNS, EvalTraceStore, ExperimentTracker, FAILURE_CLASSES, FAILURE_MODE_KIND_SPEC, FileSystemFeedbackTrajectoryStore, FileSystemRawProviderSink, FileSystemTraceStore, FindingsStore, HARBOR_IMPORT_GAP, HARNESS_NATIVE_MODEL, HeldOutGate, HoldoutAuditor, HoldoutLockedError, IMPROVEMENT_KIND_SPEC, INPUT_VALUE, INTENT_MATCH_JUDGE_VERSION, InMemoryFeedbackTrajectoryStore, InMemoryRawProviderSink, InMemoryTraceStore, InMemoryWorkspaceInspector, JudgeError, JudgeParseError, JudgeRunner, KNOWLEDGE_GAP_KIND_SPEC, KNOWLEDGE_POISONING_KIND_SPEC, LLM_CACHED_TOKENS, LLM_CACHED_TOKEN_ATTR_KEYS, LLM_CACHE_WRITE_TOKENS, LLM_CACHE_WRITE_TOKEN_ATTR_KEYS, LLM_CONTEXT_TOKENS, LLM_COST_ATTR_KEYS, LLM_COST_USD, LLM_INPUT_TOKENS, LLM_INPUT_TOKEN_ATTR_KEYS, LLM_MODEL_ATTR_KEYS, LLM_MODEL_NAME, LLM_OUTPUT_TOKENS, LLM_OUTPUT_TOKEN_ATTR_KEYS, LLM_REASONING_TOKENS, LLM_REASONING_TOKEN_ATTR_KEYS, LlmCallError, LlmClient, LlmResponseError, LlmRouteAssertionError, LockedJsonlAppender, MANN_WHITNEY_EXACT_MAX_STATES, MANN_WHITNEY_EXACT_MAX_WORK, MODEL_PRICING, MetricsCollector, ModelsUnreachableError, MultiLayerVerifier, Mutex, NoopRawProviderSink, NoopResearcher, NotFoundError, OPENINFERENCE_SPAN_KIND, OTEL_AGENT_EVAL_SCOPE, OUTPUT_VALUE, OtlpFileTraceStore, PairwiseSteeringOptimizer, ProductClient, PromptRegistry, REDACTION_VERSION, REFERENCE_EQUIVALENCE_INPUT_LIMITS, REFERENCE_EQUIVALENCE_JUDGE_VERSION, RESEARCH_REPORT_HARD_PAIR_FLOOR, ROLLOUT_SCHEMA, RUN_COST_ATTR_KEYS, ReplayCache, ReplayCacheMissError, ReplayError, RunCritic, RunIntegrityError, RunRecordValidationError, SEMANTIC_CONCEPT_JUDGE_VERSION, SKILL_USAGE_ANALYST, SPAN_KIND_ATTR_KEYS, SUPERVISOR_RUN_SCHEMA, SandboxHarness, ScenarioRegistry, SeatUnsetError, SingleBackendError, SkillUsageAnalyst, SpanNotFoundError, SubprocessSandboxDriver, TOOL_ARGS_CAPTURED, TOOL_LATENCY_MS, TOOL_NAME, TOOL_NAME_ATTR_KEYS, TRACE_ANALYST_ACTOR_DESCRIPTION, TRACE_ANALYST_ACTOR_DESCRIPTION_VERSION, TRACE_ANALYST_TRUNCATION_MARKER_PREFIX, TRACE_SCHEMA_VERSION, TokenCounter, TraceContractBuilder, TraceEmitter, TraceFileMissingError, TraceNotFoundError, UI_FINDING_SEVERITIES, UI_LENSES, UNIVERSAL_FINDERS, ValidationError, VerificationError, WILCOXON_EXACT_MAX_N, acquisitionPlansForKnowledgeGaps, agentProfileCellHashMaterial, agentProfileCellKey, agentProfileHash, agentProfileId, agentProfileModelId, agentVisibleFields, aggregateJudgeVerdicts, aggregateLlm, aggregatePrReviewScore, aggregateRunScore, allCriticalPassed, analyzeAntiSlop, analyzeSeries, analyzeSupervisorRun, analyzeSupervisorRunSources, analyzeTraces, appendScorecard, applyLlmSpanOtlpAttributes, applyToolSpanOtlpAttributes, argHash, asNumber, asString, assertCapabilityHeadroom, assertCrossFamily, assertLlmRoute, assertMinted, assertMintedLines, assertModelsServed, assertNoHiddenLeak, assertProductBenchmarkRun, assertRealAgentReceipts, assertRealBackend, assertReleaseConfidence, assertRolloutLine, assertRunAgentProfileCell, assertRunCaptured, assertSingleBackend, assignFeedbackSplit, assignHeldOutTag, attachCostToReport, attest, attributeCounterfactuals, backoffMs, deterministicSplit as benchmarkDeterministicSplit, benchmarks_exports as benchmarks, benjaminiHochberg, bisect, blendHeldout, blockingKnowledgeEval, bonferroni, bootstrapCi, buildAgentInterfaceProfileCell, buildAgentProfileCell, buildDefaultAnalystRegistry, buildDriverSystemPrompt, buildProductBenchmarkManifest, buildReflectionPrompt, buildReviewerPrompt, buildTraceAnalystTools, buildTraceInsightContext, buildTraceInsightPrompt, buildTrajectory, buildWorkerDriverSystemPrompt, byteLengthRange, cachedJudge, calibrateJudge, calibrateJudgeContinuous, callLlm, callLlmJson, canaryLeakView, canonicalJson, canonicalize, capabilityHeadroom, captureFetchToRawSink, causalAttribution, checkBehavioralCanary, checkCanaries, checkSlos, checkTraceContracts, clamp01, classifyFailure, classifyOtlpSpanRole, classifyTreatment, claudeCodeSupervisorRunReader, cliffsDelta, clusteredPairedBinary, cohensD, collectionPreserved, commentsForSource, commitBisect, comparePairedArms, compareReferenceReplay, compareToBaseline, compilerJudge, completionVerdict, composeParsers, composeValidators, computeExperimentStats, computeFindingId, computeToolUseMetrics, computeTraceMetrics, confidenceInterval, containsAll, contentHash, contextInputTokens, continuousAgreement, contractJudge, controlFailureClassFromVerification, controlRunToFeedbackTrajectory, controlRunToRunRecord, convertTraceStoresToOtlp, corpusInterRaterAgreement, corpusInterRaterAgreementFromJudgeScores, costForTokenPricing, costForUsage, costReceiptFromLlm, costReceiptFromLlmError, costReport, createAnalystAi, createAntiSlopJudge, createChatClient, createDefaultReviewer, createFeedbackTrajectory, createIntentMatchJudge, createLlmCorrectnessChecker, createLlmReviewer, createOtelExporter, createOtelTracingStore, createReferenceEquivalenceJudge, createReplayFetch, createSandboxPool, createSemanticConceptJudge, createTokenRecallChecker, createTraceAnalystKind, crossTraceDiff, crowdingDistance, dataDescriptionBits, decideNextUserTurn, decideReferenceReplayPromotion, decideReferenceReplayRunPromotion, defaultBlendWeights, defaultIsMaterial, defaultProviderRedactor, defaultReferenceReplayMatcher, defaultTraceInsightPanel, deployGateLayer, describeTraceInsightScope, diffFindings, diffScorecard, discoverPersonas, distillPlaybook, domainEvidencePattern, dominates, eProcess, ensembleJudge, errorStreakDetector, estimateCost, estimateTokens, evaluateActionPolicy, evaluateContract, evaluateHypothesis, evaluateInterimReleaseConfidence, evaluateOracles, evaluateReleaseConfidence, evaluateTraceContract, executeScenario, expandProfileAxes, expectAgent, exportProductBenchmark, exportProductBenchmarkRuns, exportRewardModel, exportRunAsOtlp, extractAssetUrls, extractErrorCount, extractOtlpAttributes, extractProducedState, extractUsage, extractUsageFromResponse, extractUsageFromSse, feedbackTrajectoriesToDatasetScenarios, feedbackTrajectoriesToOptimizerRows, feedbackTrajectoryToDatasetScenario, feedbackTrajectoryToOptimizerRow, fileContains, fileExists, fileExperimentStore, fileVerdictCache, findAutoMatchNoExpectation, findConstructorCwdDropped, findFallbackToPass, findLiteralTruePass, findProductBenchmarkArtifacts, findSkipCountsAsPass, firstNumberAttr, firstStringAttr, flattenOtlpExportToNdjson, flowLayer, fnv1a32, formatBenchmarkReport, formatDriverReport, formatFindings, formatScorecardDiff, fromHarborTrajectory, gainHistogram, gateTreatmentApplied, gateTreatmentFromMetrics, gateTreatmentFromSpans, gateTreatmentFromToolSpans, ghCliClient, gitProvenanceReader, precision as goldenPrecision, gradeOnHidden, gradeSemanticStatus, groupBy, groupRunsByAgentProfileCell, harnessAxisOf, hasCapturedToolArgs, hashContent, hashJson, hashScenarios, hashToUnit, hiddenGrade, holm, htmlContainsElement, httpGithubClient, improvementVerdict, inMemoryExperimentStore, inMemoryReferenceReplayStore, inMemoryReviewStore, inMemoryRunRecordBackend, inMemoryVerdictCache, inferDomainKeywords, inferOtlpKind, interRaterReliability, interpretCliffs, iqr, isHiddenDestination, isJudgeSpan, isLlmSpan, isModelPriced, isOtelConfigured, isOtlpModelCall, isRealnessGated, isRetrievalSpan, isRolloutLine, isRunRecord, isSandboxSpan, isToolSpan, isTrainableSplit, isTransientLlmError, isUnavailable, iterateRawCalls, jestTestParser, jsonHasKeys, jsonShape, jsonlReferenceReplayStore, jsonlReviewStore, jsonlRunRecordBackend, judgeFamily, judgeReplayGate, judgeSpans, keyPreserved, knowledgeReadinessTracePayload, leaderboard, linterJudge, llmJudge, llmSpanFromProvider, llmSpans, loadScorecard, loadScorerFromGrader, localCommandRunner, lowercaseMutator, makeEvalTools, makeFinding, makeProposalFinding, mannWhitneyU, mapConcurrent, matchGoldens, matchSpan, maximumChargeForLlmRequest, mcnemar, mcnemarPower, mcnemarRequiredN, mergeLayerResults, mergeSteeringBundle, minimumPairsForPairedDeltaTest, mintRolloutRows, modelDescriptionBits, modelHasSnapshot, modelPriceKey, mulberry32, multiToolchainLayer, noProgressDetector, normalCdf, normalizeScores, notBlocked, objectiveEval, observeAll, observedScore, observedSplitScore, otelRunCompleteHook, otlpRowsToRunRecords, otlpRowsToTraceRunRecords, otlpToRunRecords, otlpToTraceRunRecords, pairArms, pairRunRecords, pairedBootstrap, pairedCohensDz, pairedDeltaTest, pairedEvalueSequence, pairedMde, pairedRiskDifference, pairedSignTest, pairedTTest, paraphraseRobustness, paraphraseRobustnessScenarios, paretoChart, paretoFrontier, paretoFrontierWithCrowding, parseCorrectnessResponse, parseFeedbackTrajectoriesJsonl, parseReflectionResponse, parseRunRecordSafe, parseRuntimeTrajectoryHookEvent, partialCredit, partitionHeldOut, passAtK, passOrthogonality, pearsonR, pixelDeltaRatio, planTraceInsightQuestions, politenessPrefixMutator, positionalBias, preflightModels, printDriverSummary, probeLlm, productBenchmarkIntegrityFailures, productBenchmarkMutableSurfaces, productBenchmarkRepoIdentity, productBenchmarkSplits, profile_exports as profile, projectOtlpFlatLine, projectRuntimeTrajectoryEvidence, promptBisect, proposeSynthesisTargets, providerFromBaseUrl, pytestTestParser, ranks, readClaudeCodeSupervisorRun, readOtlpStatus, readProductBenchmarkManifest, readProductBenchmarkRecords, recordRuns, recordRunsToScorecard, redTeamDataset, redTeamReport, redactString, redactValue, referenceReplayRunsToSteeringRows, referenceReplayScenarioToRunScore, regexMatch, regexMatches, relabelImportedSplit, renderMarkdownReport, renderPlaybookMarkdown, renderPreferenceMemoryMarkdown, renderPriorFindings, renderReleaseReport, renderSteeringText, renderSupervisorRunHeadline, renderSupervisorRunMarkdown, renderUpstreamFindings, repeatedActionDetector, replayFeedbackTrajectories, replayFeedbackTrajectory, replayScorerOverCorpus, replayTraceThroughJudge, requireAgentProfileCell, requiredPairedSampleSize, requiredSampleSize, researchReport, resolveModelPricing, resolveSeat, rollupSupervisorRuns, roundTripRunRecord, routeFields, rowCount, rowWhere, runAgentControlLoop, runAssertions, runBehavioralCanaries, runCanaries, runCounterfactual, runE2EWorkflow, runEvalCampaign, runExpectations, runFailureClass, runHarnessExperiment, runIntentMatchJudge, runJudgeFleet, runKeywordCoverageJudge, runKeywordCoverageJudgeUrl, runLiveProof, runProposeReview, runProposeReviewAsControlLoop, runRecordToProductBenchmarkRecord, runReferenceEquivalenceJudge, runReferenceReplay, runScore, runSelfPlay, runSemanticConceptJudge, runTaskScore, runTestGradedScenario, runsForScenario, scalarScore, scanForMuffledGates, scoreContinuity, scoreFromEvals, scoreKnowledgeReadiness, scoreOrigin, scorePrReviewComments, scorePrReviewSource, scoreRedTeamOutput, scoreReferenceReplay, scoreTraceInsightReadiness, seatPresets, securityJudge, selectHarnessVariant, selfPreference, sentenceReorderMutator, serializeFeedbackTrajectoriesJsonl, showMeasured, signManifest, spearmanR, statusAdvanced, stopOnNoProgress, stopOnRepeatedAction, stringField, stripFencedJson, studentTCdf, subjectiveEval, summarizeAgentReceiptIntegrity, summarizeBackendIntegrity, summarizeHarnessResults, summarizePrReviewBenchmark, summarizePreferenceMemory, summaryTable, supervisorRunRolloutLines, testJudge, textInSnapshot, throwIfRunIncomplete, toAgentProfileJson, toHarborTrajectories, toHarborTrajectory, toJsonl, toLangfuseEnvelope, toOpenAiTool, toPrometheusText, toRewardRows, toSftRows, tokenizeDomainWords, toolNamesForRun, toolSpans, traceAnalystFunctionGroup, traceAnalystOnRunComplete, traceContract, traceJudge, traceJudgeEnsemble, traceSpanKindToOpenInferenceKind, tracedAnalyzeTraces, trainingReward, trainingScore, typoMutator, urlContains, userQuestionsForKnowledgeGaps, validateAgentProfileCell, validateProductBenchmarkManifest, validateProductBenchmarkRecord, validateProductBenchmarkRun, validateRolloutLine, validateRunRecord, verbosityBias, verifyAgentProfileCell, verifyAttestation, verifyCompletion, verifyManifest, visualDiff, viteDeployRunner, vitestTestParser, weightedComposite, weightedMean, weightedRecall, welchsTTest, whitespaceCollapseMutator, wilcoxonSignedRank, wilson, withAssignedFeedbackSplit, withHeldoutBlend, withJudgeRetry, withOtelPipeline, wranglerDeployRunner, writeSupervisorRunReport };
|
|
12127
12297
|
|
|
12128
12298
|
//# sourceMappingURL=index.js.map
|