@tangle-network/agent-eval 0.114.0 → 0.115.1

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/dist/index.d.ts CHANGED
@@ -16,14 +16,13 @@ import { F as FailureClass, T as ToolSpan, h as BudgetSpec, B as BudgetLedgerEnt
16
16
  export { A as Artifact, E as EventKind, i as FAILURE_CLASSES, G as GenericSpan, J as JudgeSpan, M as Message, c as RetrievalSpan, g as RunLayer, f as RunStatus, d as SandboxSpan, j as SpanBase, b as SpanKind, k as SpanStatus, l as TRACE_SCHEMA_VERSION, e as TraceEvent, m as isJudgeSpan, n as isLlmSpan, o as isRetrievalSpan, p as isSandboxSpan, q as isToolSpan } from './schema-SGWcK9wa.js';
17
17
  import { A as AgentEvalError, J as JudgeError, a as ConfigError } from './errors-oeQrLqXC.js';
18
18
  export { b as AgentEvalErrorCode, C as CaptureIntegrityError, N as NotFoundError, R as ReplayError, V as ValidationError, c as VerificationError } from './errors-oeQrLqXC.js';
19
- import { b as CorrectnessChecker } from './pre-registration-CTQbZbpX.js';
20
- export { A as ArtifactCheckArtifact, d as ArtifactEventLike, e as ArtifactValidator, f as BackendIntegrityError, B as BackendIntegrityReport, C as CompletionRequirement, a as CompletionVerdict, H as HypothesisManifest, g as HypothesisResult, h as LlmCorrectnessCheckerOpts, L as LlmJudgeDimension, c as LlmJudgeOptions, i as ProducedProposal, P as ProducedState, j as ProposalEventLike, k as RequirementCheck, R as RuntimeEventLike, m as SatisfiedBy, S as SignedManifest, n as SignedManifestAlgo, T as TaskGold, o as ToolCallEventLike, V as ValidationContext, p as ValidationIssue, q as ValidationResult, r as assertRealBackend, s as byteLengthRange, t as canonicalize, u as completionVerdict, v as composeValidators, w as containsAll, x as createLlmCorrectnessChecker, y as createTokenRecallChecker, z as evaluateHypothesis, D as extractProducedState, E as hashJson, F as jsonHasKeys, l as llmJudge, G as parseCorrectnessResponse, I as regexMatch, J as signManifest, K as summarizeBackendIntegrity, M as verifyCompletion, N as verifyManifest } from './pre-registration-CTQbZbpX.js';
19
+ import { c as CorrectnessChecker } from './pre-registration-oNItiRBb.js';
20
+ export { A as ArtifactCheckArtifact, e as ArtifactEventLike, f as ArtifactValidator, g as BackendIntegrityError, B as BackendIntegrityReport, h as ComparePairedArmsOptions, C as CompletionRequirement, a as CompletionVerdict, H as HypothesisManifest, i as HypothesisResult, j as LlmCorrectnessCheckerOpts, L as LlmJudgeDimension, d as LlmJudgeOptions, M as MatchedPair, k as PairArmsOptions, m as PairArmsResult, n as PairedArmRow, P as PairedArmsComparison, o as PairedCorrectness, p as PairedMetricDelta, q as ProducedProposal, b as ProducedState, r as ProposalEventLike, s as RequirementCheck, R as RuntimeEventLike, t as SatisfiedBy, S as SignedManifest, u as SignedManifestAlgo, T as TaskGold, v as ToolCallEventLike, V as ValidationContext, w as ValidationIssue, x as ValidationResult, y as assertRealBackend, z as byteLengthRange, D as canonicalize, E as comparePairedArms, F as completionVerdict, G as composeValidators, I as containsAll, J as createLlmCorrectnessChecker, K as createTokenRecallChecker, N as evaluateHypothesis, O as extractProducedState, Q as hashJson, U as jsonHasKeys, l as llmJudge, W as pairArms, X as parseCorrectnessResponse, Y as regexMatch, Z as signManifest, _ as summarizeBackendIntegrity, $ as verifyCompletion, a0 as verifyManifest } from './pre-registration-oNItiRBb.js';
21
21
  import { T as TraceEmitter } from './emitter-BRchAAAx.js';
22
22
  export { R as RunCompleteHook, a as RunCompleteHookContext, S as SpanHandle, b as TraceEmitterOptions, l as llmSpanFromProvider } from './emitter-BRchAAAx.js';
23
23
  import { h as ReleaseConfidenceThresholds, f as ReleaseConfidenceScorecard } from './release-report-oBfOz8ku.js';
24
24
  export { A as ActionableSideInfo, o as AsiSeverity, B as BootstrapOptions, a as BootstrapResult, J as JudgeReplayGateArgs, R as ReleaseConfidenceAxis, b as ReleaseConfidenceAxisName, c as ReleaseConfidenceInput, d as ReleaseConfidenceIssue, e as ReleaseConfidenceMetrics, g as ReleaseConfidenceStatus, i as ReleaseTraceEvidence, j as RenderReleaseReportOptions, V as Verdict, k as assertReleaseConfidence, l as bootstrapCi, m as evaluateReleaseConfidence, n as judgeReplayGate, r as renderReleaseReport } from './release-report-oBfOz8ku.js';
25
- import { P as PairedBootstrapOptions, M as McNemarResult, R as RiskDifferenceResult, a as PairedBootstrapResult } from './statistics-oUbOJe-S.js';
26
- export { c as CliffsMagnitude, d as CorpusAgreementOptions, e as CorpusAgreementPerDimension, C as CorpusAgreementReport, f as CorpusScoreRecord, g as EProcess, h as EProcessOptions, E as EProcessState, i as EProcessStep, j as PairedSignTestResult, k as ProportionInterval, S as SignTestAlternative, W as WeightedCompositeInput, l as WeightedCompositeResult, b as benjaminiHochberg, m as bonferroni, n as cliffsDelta, o as cohensD, q as confidenceInterval, r as corpusInterRaterAgreement, s as corpusInterRaterAgreementFromJudgeScores, t as eProcess, u as holm, v as interRaterReliability, x as interpretCliffs, y as mannWhitneyU, z as mcnemar, A as mcnemarPower, B as mcnemarRequiredN, D as mulberry32, F as normalizeScores, p as pairedBootstrap, G as pairedMde, H as pairedRiskDifference, I as pairedSignTest, J as pairedTTest, K as partialCredit, L as passAtK, N as pearsonR, O as ranks, Q as requiredSampleSize, T as spearmanR, U as weightedComposite, V as weightedMean, w as wilcoxonSignedRank, X as wilson } from './statistics-oUbOJe-S.js';
25
+ export { c as CliffsMagnitude, d as CorpusAgreementOptions, e as CorpusAgreementPerDimension, C as CorpusAgreementReport, f as CorpusScoreRecord, g as EProcess, h as EProcessOptions, E as EProcessState, i as EProcessStep, M as McNemarResult, P as PairedBootstrapOptions, a as PairedBootstrapResult, j as PairedSignTestResult, k as ProportionInterval, R as RiskDifferenceResult, S as SignTestAlternative, W as WeightedCompositeInput, l as WeightedCompositeResult, b as benjaminiHochberg, m as bonferroni, n as cliffsDelta, o as cohensD, q as confidenceInterval, r as corpusInterRaterAgreement, s as corpusInterRaterAgreementFromJudgeScores, t as eProcess, u as holm, v as interRaterReliability, x as interpretCliffs, y as mannWhitneyU, z as mcnemar, A as mcnemarPower, B as mcnemarRequiredN, D as mulberry32, F as normalizeScores, p as pairedBootstrap, G as pairedMde, H as pairedRiskDifference, I as pairedSignTest, J as pairedTTest, K as partialCredit, L as passAtK, N as pearsonR, O as ranks, Q as requiredSampleSize, T as spearmanR, U as weightedComposite, V as weightedMean, w as wilcoxonSignedRank, X as wilson } from './statistics-oUbOJe-S.js';
27
26
  import { OtelExporter, OtelExportConfig } from './traces.js';
28
27
  export { CaptureFetchContext, CaptureFetchOptions, DEFAULT_REDACTION_RULES, ExportableSpan, ExtractedUsage, FlattenOtlpOptions, LLM_CACHED_TOKENS, LLM_CACHED_TOKEN_ATTR_KEYS, LLM_COST_ATTR_KEYS, LLM_COST_USD, LLM_INPUT_TOKENS, LLM_INPUT_TOKEN_ATTR_KEYS, LLM_MODEL_ATTR_KEYS, LLM_MODEL_NAME, LLM_OUTPUT_TOKENS, LLM_OUTPUT_TOKEN_ATTR_KEYS, OPENINFERENCE_SPAN_KIND, OTEL_AGENT_EVAL_SCOPE, OtlpExport, OtlpFileTraceStore, OtlpFileTraceStoreOptions, OtlpFlatLine, OtlpResourceSpans, OtlpSpan, OtlpToRunRecordsOptions, OtlpTraceRunRecord, ProjectedOtlpSpan, REDACTION_VERSION, RedactionReport, RedactionRule, ReplayCache, ReplayCacheEntry, ReplayCacheMissError, ReplayCacheStats, ReplayFetchOptions, SPAN_KIND_ATTR_KEYS, SpanNotFoundError, TOOL_NAME, TOOL_NAME_ATTR_KEYS, TRACE_ANALYST_ACTOR_DESCRIPTION, TRACE_ANALYST_ACTOR_DESCRIPTION_VERSION, TRACE_ANALYST_SUBAGENT_DESCRIPTION, TraceAggregate, TraceAnalystHookOptions, TraceFileMissingError, TraceInsightContext, TraceInsightFinding, TraceInsightPanelRole, TraceInsightPromptInput, TraceInsightQualityGate, TraceInsightQuestion, TraceInsightReadiness, TraceInsightSuite, TraceInsightTask, TraceNotFoundError, TraceStoreSource, TraceStoreToOtlpOptions, TracesToOtlpResult, asNumber, asString, buildTraceAnalystTools, buildTraceInsightContext, buildTraceInsightPrompt, captureFetchToRawSink, convertTraceStoresToOtlp, createOtelExporter, createOtelTracingStore, createReplayFetch, defaultTraceInsightPanel, describeTraceInsightScope, domainEvidencePattern, exportRunAsOtlp, extractOtlpAttributes, extractUsage, extractUsageFromResponse, extractUsageFromSse, firstNumberAttr, firstStringAttr, flattenOtlpExportToNdjson, inferDomainKeywords, inferOtlpKind, iterateRawCalls, otelRunCompleteHook, otlpToRunRecords, otlpToTraceRunRecords, planTraceInsightQuestions, projectOtlpFlatLine, readOtlpStatus, redactString, redactValue, scoreTraceInsightReadiness, stringField, tokenizeDomainWords, traceAnalystFunctionGroup, traceAnalystOnRunComplete, traceSpanKindToOpenInferenceKind } from './traces.js';
29
28
  import { a as AnalyzeTracesInput, A as AnalyzeTracesOptions, b as AnalyzeTracesResult } from './analyst-C8HHvfJp.js';
@@ -1154,159 +1153,6 @@ declare class MetricsCollector {
1154
1153
  getConvergenceCurve(): number[];
1155
1154
  }
1156
1155
 
1157
- /**
1158
- * Matched-pair arm comparison — "did the treatment arm beat the baseline arm
1159
- * on the SAME work items?"
1160
- *
1161
- * An arm A/B over run records is only trustworthy when it is PAIRED: the same
1162
- * task/scenario/seed evaluated under both arms, compared item-by-item, so
1163
- * inter-item difficulty variance cancels instead of masquerading as an arm
1164
- * effect. This module owns the two error-prone steps every consumer otherwise
1165
- * hand-rolls:
1166
- *
1167
- * 1. Pairing — matching rows across arms by `pairKey` (and by `repKey`
1168
- * within multi-rep items), with leftovers REPORTED rather than silently
1169
- * dropped (a silently unbalanced pairing biases every paired statistic
1170
- * downstream). Pairing never keys on outcome content: matching reps by
1171
- * their outcomes deflates discordant-pair counts and makes McNemar
1172
- * anti-conservative, so reps pair only by (`pairKey`, `repKey`) identity.
1173
- * 2. Composition — feeding the matched pairs to the correct paired
1174
- * estimators that already live in `statistics`: `mcnemar` +
1175
- * `pairedRiskDifference` for pass/fail, `pairedBootstrap` +
1176
- * `wilcoxonSignedRank` for continuous metrics. No statistic is
1177
- * re-implemented here.
1178
- *
1179
- * The row shape is deliberately structural — callers project a `RunRecord`
1180
- * (or any record) into `{ pairKey, arm, pass?, metrics? }`. Arm names are
1181
- * caller-supplied parameters; the module ships no domain literal.
1182
- */
1183
-
1184
- /** One arm observation of one work item. Structural on purpose: callers
1185
- * project their own record type (e.g. a `RunRecord`) into this shape. */
1186
- interface PairedArmRow {
1187
- /** Matching key — rows sharing a `pairKey` across both arms form pairs
1188
- * (typically the task/scenario/seed identity). */
1189
- pairKey: string;
1190
- /** Rep identity within a `pairKey` (e.g. a seed or rep number). Required on
1191
- * every row of a `pairKey` that has more than one rep in either arm; reps
1192
- * then pair only on exact (`pairKey`, `repKey`) match, never on outcome
1193
- * content. Optional when each arm has at most one rep of the item. */
1194
- repKey?: string;
1195
- /** Arm label this row was produced under. */
1196
- arm: string;
1197
- /** Binary outcome; omit when the comparison has no pass/fail notion. */
1198
- pass?: boolean;
1199
- /** Named numeric measurements (score, cost, latency, …). */
1200
- metrics?: Record<string, number>;
1201
- }
1202
- interface PairArmsOptions {
1203
- /** Arm treated as the control side of every pair. */
1204
- baselineArm: string;
1205
- /** Arm treated as the treatment side of every pair. */
1206
- treatmentArm: string;
1207
- }
1208
- /** One matched (baseline, treatment) observation of the same work item. */
1209
- interface MatchedPair {
1210
- pairKey: string;
1211
- /** 0-based position of this pair within its `pairKey`, ordered by sorted
1212
- * `repKey` (always 0 for a single-rep item). The rep identity itself is on
1213
- * the rows (`baseline.repKey` / `treatment.repKey`). */
1214
- repIndex: number;
1215
- baseline: PairedArmRow;
1216
- treatment: PairedArmRow;
1217
- }
1218
- interface PairArmsResult {
1219
- /** Matched pairs, ordered by (`pairKey`, `repIndex`). */
1220
- pairs: MatchedPair[];
1221
- /** Baseline rows left without a treatment counterpart — reported, never
1222
- * silently dropped. */
1223
- unpairedBaseline: PairedArmRow[];
1224
- /** Treatment rows left without a baseline counterpart. */
1225
- unpairedTreatment: PairedArmRow[];
1226
- }
1227
- /**
1228
- * Match rows across two arms into (baseline, treatment) pairs by `pairKey`.
1229
- *
1230
- * A `pairKey` with at most one row per arm pairs directly, no `repKey`
1231
- * needed. A `pairKey` with multiple reps in either arm requires `repKey` on
1232
- * every one of its rows, and reps pair only on exact (`pairKey`, `repKey`)
1233
- * match — pairing is keyed purely on row identity, never on outcome content
1234
- * (outcome-keyed matching deflates discordant counts and biases McNemar), and
1235
- * is therefore independent of input order. Reps whose `repKey` has no
1236
- * counterpart in the other arm, and items present in only one arm, land in
1237
- * the unpaired lists — reported, never truncated.
1238
- *
1239
- * Fail-loud: throws when either named arm has zero rows (an unknown arm
1240
- * name would otherwise read as "everything unpaired"), when the two arm
1241
- * names are equal, when a multi-rep `pairKey` has a row without `repKey`, or
1242
- * when a (`pairKey`, arm) group repeats a `repKey` (the match would be
1243
- * ambiguous).
1244
- */
1245
- declare function pairArms(rows: readonly PairedArmRow[], opts: PairArmsOptions): PairArmsResult;
1246
- /** Paired pass/fail comparison over the pairs where BOTH sides carry `pass`. */
1247
- interface PairedCorrectness {
1248
- /** Discordant pairs where the treatment passed and the baseline failed. */
1249
- b10: number;
1250
- /** Discordant pairs where the baseline passed and the treatment failed. */
1251
- b01: number;
1252
- /** Exact McNemar significance over the paired outcomes (`b === b10`, `c === b01`). */
1253
- mcnemar: McNemarResult;
1254
- /** Paired effect size: p(treatment) − p(baseline) with a paired-variance CI. */
1255
- riskDifference: RiskDifferenceResult;
1256
- }
1257
- /** Paired delta summary for one named metric (delta = treatment − baseline). */
1258
- interface PairedMetricDelta {
1259
- name: string;
1260
- /** Pairs where BOTH sides carry a finite value for this metric. */
1261
- n: number;
1262
- /** Pairs where at least one side does not carry the metric. */
1263
- nMissing: number;
1264
- /** Median paired delta; NaN when `n === 0` (no data ≠ measured zero). */
1265
- medianDelta: number;
1266
- /** Mean paired delta; NaN when `n === 0`. */
1267
- meanDelta: number;
1268
- /** Bootstrap CI on the paired delta (`pairedBootstrap`); null when
1269
- * `n === 0` — a zero-width [0, 0] interval on no data would read as a
1270
- * measured tight null. */
1271
- bootstrapCi: PairedBootstrapResult | null;
1272
- /** Wilcoxon signed-rank test on the paired deltas; null when `n === 0`. */
1273
- wilcoxon: {
1274
- w: number;
1275
- p: number;
1276
- } | null;
1277
- }
1278
- interface ComparePairedArmsOptions extends PairArmsOptions {
1279
- /** Metrics to compare. Default: every metric name observed on any matched
1280
- * pair, sorted. A name that appears on no pair is still reported (with
1281
- * `n = 0`) so a misspelled metric is visible instead of vanishing. */
1282
- metricNames?: string[];
1283
- /** Passed through to `pairedBootstrap` — set `seed` for reproducible CIs. */
1284
- bootstrap?: PairedBootstrapOptions;
1285
- }
1286
- interface PairedArmsComparison {
1287
- nPairs: number;
1288
- nUnpairedBaseline: number;
1289
- nUnpairedTreatment: number;
1290
- /** null when no matched pair carries `pass` on both sides — a pass/fail
1291
- * verdict over rows that never measured pass/fail would be fabricated. */
1292
- correctness: PairedCorrectness | null;
1293
- metricDeltas: PairedMetricDelta[];
1294
- }
1295
- /**
1296
- * Full matched-pair arm comparison: pair via {@link pairArms}, then compose
1297
- * the paired estimators from `statistics` over the matched pairs.
1298
- *
1299
- * Correctness uses only the pairs where both sides carry `pass` (`mcnemar.n`
1300
- * is that subset's size); each metric uses only the pairs where both sides
1301
- * carry a finite value for it, with the remainder counted in `nMissing`.
1302
- * Deltas are treatment − baseline throughout.
1303
- *
1304
- * Fail-loud: inherits {@link pairArms}'s unknown-arm throw, and throws on a
1305
- * non-finite metric value — silently treating corrupt telemetry as "metric
1306
- * absent" would misreport it as missing coverage.
1307
- */
1308
- declare function comparePairedArms(rows: readonly PairedArmRow[], opts: ComparePairedArmsOptions): PairedArmsComparison;
1309
-
1310
1156
  type PrReviewSource = 'drew' | 'donovan' | 'shady' | 'codex' | 'claude-code' | 'gpt-5.5-high' | 'claude-opus-4.7-high' | 'kimi' | 'opencode' | (string & {});
1311
1157
  type PrReviewSeverity = 'critical' | 'high' | 'medium' | 'low' | 'nit';
1312
1158
  type PrReviewOutcome = 'accepted' | 'fixed' | 'rejected' | 'duplicate' | 'noise' | 'unknown';
@@ -7035,4 +6881,4 @@ type CachedJudge<TArtifact, TScenario extends Scenario$1 = Scenario$1> = JudgeCo
7035
6881
  */
7036
6882
  declare function cachedJudge<TArtifact, TScenario extends Scenario$1 = Scenario$1>(judge: JudgeConfig<TArtifact, TScenario>, store: VerdictCacheStore, options: CachedJudgeOptions): CachedJudge<TArtifact, TScenario>;
7037
6883
 
7038
- export { ATTESTATION_ALGORITHM, type ActiveLearningOptions, type AdapterRun, AgentDriver, type AgentDriverConfig, AgentEvalError, type AgentProfileRuntimeReceipt, type AgreementResult, type AlignmentOp, AnalyzeTracesInput, AnalyzeTracesOptions, AnalyzeTracesResult, type AntiSlopConfig, type AntiSlopIssue, type AntiSlopReport, type AssertCapabilityHeadroomOptions, type AssertCrossFamilyOptions, type AssertSingleBackendOptions, type AttestationProvenance, type AttestationVerification, type AttestedReport, type AutoPrClient, AxGepaSteeringOptimizer, type AxSteeringOptimizerConfig, type BackendDescriptor, BaselineReport, BehaviorAssertion, BehavioralMetrics, BenchmarkReport, BenchmarkRunner, BenchmarkRunnerConfig, type BisectOptions, type BisectResult, type BisectStep, type BlendWeights, BudgetBreachError, BudgetGuard, BudgetLedgerEntry, BudgetSpec, type BuildAgreementJudgeOptions, CODING_HARNESSES, type CachedJudge, type CachedJudgeOptions, CallExpectation, type CanaryAlert, type CanaryKind, type CanaryLeak, type CanaryOptions, type CanaryReport, type CanarySeverity, type CandidateComparison, type CandidateScenario, type CapabilityHeadroomOptions, type CapabilityHeadroomResult, type CausalAttributionReport, type CellVerdict, ChannelRollup, ChatRequest, CheckResult, type ClusterBootstrapInterval, type ClusterSignFlipAlternative, type ClusterSignFlipResult, type ClusteredBinaryCluster, type ClusteredMatchedPair, type ClusteredPairedBinaryOptions, type ClusteredPairedBinaryResult, type ClusteredPairedBinaryStatistics, CollectedArtifacts, type CommandRunner, type CompareLabels, type ComparePairedArmsOptions, CompletionCriterion, ConfigError, type ContinuityCheck, type ContinuityCheckResult, type ContinuityReport, type ContinuitySnapshotPair, type ContractCheckResult, type ContractJudgeOptions, type ContractMetric, type ContractReport, type ContractRule, type ContractRuleKind, type ContractSpan, type ContractVerdict, type ContractViolation, ControlEvalResult, ControlSeverity, ConvergenceTracker, CorrectnessChecker, type CostEntry, CostLedger, type CostReport, type CostSummary, CostTracker, type CounterfactualContext, type CounterfactualMutation, type CounterfactualResult, type CounterfactualRunner, CreateChatClientOpts, type CreateDefaultReviewerOptions, type CreateExperimentInput, type CreateSandboxPoolOpts, CrossFamilyError, type CrossTraceDiff, type CrossTraceDiffOptions, DEFAULT_AGENT_SLOS, DEFAULT_FINDERS, DEFAULT_HARNESS_OBJECTIVES, DEFAULT_MUTATION_PRIMITIVES, DEFAULT_MUTATORS, DEFAULT_PR_REVIEW_SCORE_WEIGHTS, DEFAULT_SEVERITY_WEIGHTS, type DataAcquisitionPlan, Dataset, DatasetScenario, type DecideNextUserTurnOpts, DefaultVerdict, type DeployFamily, type DeployGateLayerInput, type DeployRunResult, type DeployRunner, type DescriptionLengthCandidate, type DescriptionLengthConfig, type DescriptionLengthDecision, type DescriptionLengthEvidence, DescriptionLengthGate, type DescriptionLengthRejectionCode, type DetectorEvent, type DetectorSeverity, type DetectorSignal, type DiffScorecardOptions, type DirEntry, type DiscoverPersonasOptions, type DiscoveredPersona, DriverResult, DriverState, DualAgentBench, type DualAgentBenchConfig, type DualAgentReport, type DualAgentRound, type DualAgentScenario, type DualAgentScenarioResult, ERROR_COUNT_PATTERNS, type EnsembleAggregate, type EnsembleJudgeOptions, type ErrorCountPattern, type ErrorStreakOptions, type EvalToolDef, EvalTraceStore, type EvolutionRound, type ExecutorConfig, type Expectation, type Experiment, type ExperimentProvenance, type ExperimentRep, type ExperimentStats, type ExperimentStore, ExperimentTracker, type ExperimentTrackerOptions, type ExperimentVerdict, type ExportedRewardModel, type ExtractOptions, type ExtractResult, type FactorContribution, type FactorialCell, FailureClass, FeedbackLabel, FeedbackTrajectory, FeedbackTrajectoryStore, type FieldAgreementSpec, type FieldDestination, type FileChange, type FlowAction, type FlowLayerEnv, type FlowLayerFactoryInput, type FlowRunner, type FlowRunnerStepResult, type FlowSpec, type FlowStep, type GhCliClientOptions, type GoldScenario, type GoldSplit, type GoldenSeverity, type GoldenSpec, HARNESS_NATIVE_MODEL, type HarnessAdapter, HarnessConfig, type HarnessExperimentConfig, type HarnessExperimentResult, type HarnessIntervention, type HarnessRunRequest, type HarnessRunResult, type HarnessScenario, type HarnessSelection, type HarnessVariant, type HarnessVariantReport, type HeadroomClass, type HeadroomInput, type HeldOutPartition, type HiddenCriteriaGrader, type HiddenGradeResult, type HiddenLeak, HoldoutAuditor, type HttpGithubClientOptions, INTENT_MATCH_JUDGE_VERSION, type ImageData, type ImprovementThresholds, type ImprovementVerdictResult, InMemoryWorkspaceInspector, type InferenceScorer, type InspectorContext, type IntentMatchInput, type IntentMatchOptions, type IntentMatchResult, type InteractionContribution, JudgeError, type JudgeFamily, type JudgeFleetOptions, JudgeFn, JudgeParseError, type JudgeReplayResult, type JudgeRetryOutcome, type JudgeRetryPolicy, JudgeRunner, type JudgeScoreInput, type JudgeVerdict, type KeywordConceptSpec, type KeywordCoverageFinding, type KeywordCoverageOptions, type KeywordCoverageResult, type KnowledgeAcquisitionMode, type KnowledgeBundle, type KnowledgeFallbackPolicy, type KnowledgeFreshness, type KnowledgeImportance, type KnowledgeReadinessReport, type KnowledgeRecommendedAction, type KnowledgeRequirement, type KnowledgeRequirementCategory, type KnowledgeResponsibleSurface, type KnowledgeSensitivity, type LangfuseEnvelope, type LangfuseGeneration, type LangfuseScore, Layer, LayerResult, type LeaderboardOptions, type LeaderboardRow, type LiveProofArtifact, type LiveProofConfig, type LiveProofContext, type LiveProofResult, LlmClientOptions, LlmSpan, LockedJsonlAppender, MODEL_PRICING, type MakeEvalToolsConfig, type MatchResult, type MatchedPair, type MatcherResult, McNemarResult, type MeasurementPolicy, type MergeOptions, MetricsCollector, type ModelCostRollup, type ModelPreflight, type ModelSeats, ModelsUnreachableError, type MuffledFinder, type MuffledFinding, type MultiToolchainLayerConfig, type Mutator, Mutex, type NoLeakOptions, type NoProgressOptions, Objective, type Oracle, type OracleObservation, type OracleReport, type OracleResult, type OrthogonalityInput, type OrthogonalityResult, OtelExportConfig, OtelExporter, type OtelPipelineHandle, type OtelPipelineOptions, type PairArmsOptions, type PairArmsResult, type PairedArmRow, type PairedArmsComparison, PairedBootstrapOptions, PairedBootstrapResult, type PairedCorrectness, type PairedMetricDelta, PairwiseSteeringOptimizer, type ParaphraseRobustnessScenarioInput, type ParaphraseRobustnessScenarioResult, ParetoResult, type ParseStudentLabel, type PartitionHeldOutOptions, PersonaConfig, type Playbook, type PlaybookEntry, type PoolSlot, type PrReviewAuditCase, type PrReviewBenchmarkSummary, type PrReviewComment, type PrReviewMatchedFinding, type PrReviewOutcome, type PrReviewReferenceFinding, type PrReviewScore, type PrReviewScoreWeights, type PrReviewSeverity, type PrReviewSource, type PreflightModelsOptions, type PreflightOutcome, type ProductBenchmarkArm, type ProductBenchmarkArtifactPaths, type ProductBenchmarkBudgets, type ProductBenchmarkExportOptions, type ProductBenchmarkExportResult, type ProductBenchmarkManifest, type ProductBenchmarkProfileRef, type ProductBenchmarkRecord, type ProductBenchmarkRepoRef, type ProductBenchmarkRunInput, type ProductBenchmarkScenario, type ProductBenchmarkSingleRunExportOptions, type ProductBenchmarkSplit, type ProductBenchmarkSubstrateVersions, type ProductBenchmarkValidationReport, ProductClient, ProductClientConfig, type ProfileAxisSpec, type PromptHandle, PromptRegistry, type ProposeAutomatedPullRequestInput, type ProposeAutomatedPullRequestResult, type ProvenanceReader, type RecordRunsOptions, type ReferenceMatchResult, type ReferenceReplayAdapter, type ReferenceReplayAdapterFn, type ReferenceReplayAdapterLike, type ReferenceReplayAggregate, type ReferenceReplayCandidate, type ReferenceReplayCase, type ReferenceReplayCaseRun, type ReferenceReplayExecutionScenario, type ReferenceReplayItem, type ReferenceReplayMatch, type ReferenceReplayMatchStrategy, type ReferenceReplayMatcher, type ReferenceReplayPromotionDecision, type ReferenceReplayPromotionPolicy, type ReferenceReplayRun, type ReferenceReplayRunContext, type ReferenceReplayRunOptions, type ReferenceReplayRunStore, type ReferenceReplayScenario, type ReferenceReplayScenarioScore, type ReferenceReplayScore, type ReferenceReplayScoreOptions, type ReferenceReplaySplit, type ReferenceReplaySplitComparison, type ReferenceReplaySteeringRowsOptions, type ReflectionContext, type ReflectionProposal, ReleaseConfidenceScorecard, ReleaseConfidenceThresholds, type RenderStudentPrompt, type RepeatedActionOptions, type RepoRef, type ReviewerMemoryEntry, type ReviewerOutput, type ReviewerPromptInput, type ReviewerSoftFailDefaults, type ReviewerVerificationSummary, RiskDifferenceResult, type RobustnessResult, type RoutedField, Run, type RunCommandInput, type RunCommandResult, type RunDistillationOptions, type RunDistillationResult, RunFilter, RunRecord, type RunRecordBackend, type RunRecordFilter, RunScore, RunScoreWeights, RunSplitTag, RunTrace, type RuntimeResolution, SandboxDriver, SandboxHarnessResult, type SandboxJudgeKind, type SandboxJudgeResult, type SandboxJudgeSpec, type SandboxPool, type ScanOptions, Scenario, type ScenarioCost, ScenarioFile, ScenarioRegistry, ScenarioResult, type ScoreKnowledgeReadinessOptions, type Scorecard, type ScorecardCell, type ScorecardCellDiff, type ScorecardDiff, type ScorecardEntry, type ScorecardLogLine, type ScoredTarget, type SeatName, type SeatPresetName, SeatUnsetError, type SelfPlayOptions, type SelfPlayProposer, type SelfPlayScorer, type SerializedRegex, Severity, type SingleBackendDivergence, SingleBackendError, type SingleBackendField, type SingleBackendReport, type Slo, type SloCheckResult, type SloComparator, type SloReport, type SloSeverity, type SlopCategory, type SlotFactory, Span, type SpanPredicate, type SplitGoldOptions, type SteeringBundle, type SteeringDelta, type SteeringOptimizationResult, type SteeringOptimizationRow, type SteeringOptimizationSelector, type SteeringOptimizerBackend, type SteeringOptimizerConfig, type SteeringRolePrompt, type StepAttribution, type StreamingDetector, type SynthesisReason, type SynthesisTarget, type TaskHeadroom, TestResult, type TextMatcher, type ThresholdContract, TokenCounter, type TokenSpec, type ToolMatcher, ToolSpan, TraceAnalystSpan, type TraceContract, TraceContractBuilder, TraceEmitter, TraceStore, type TracedAnalystOptions, type TracedJudgeOptions, Trajectory, TrajectoryStep, type TreatmentClass, type TreatmentGate, type TreatmentGateInput, type TreatmentGateOptions, type TrialTrace, TurnMetrics, UI_FINDING_SEVERITIES, UI_LENSES, UNIVERSAL_FINDERS, type UiFinding, type UiFindingScreenshot, type UiFindingSeverity, type UiLens, type UserQuestion, type VerdictCacheStats, type VerdictCacheStore, VerifyContext, type VisualDiffOptions, type VisualDiffResult, type ViteDeployRunnerInput, type WorkerDriverContext, type WorkflowTopology, type WorkspaceAssertion, type WorkspaceAssertionResult, type WorkspaceInspector, type WorkspaceSnapshot, type WranglerDeployRunnerInput, acquisitionPlansForKnowledgeGaps, adversarialJudge, agentProfileHash, agentProfileId, agentProfileModelId, agentVisibleFields, aggregateJudgeVerdicts, aggregatePrReviewScore, analyzeAntiSlop, appendScorecard, assertCapabilityHeadroom, assertCrossFamily, assertModelsServed, assertNoHiddenLeak, assertProductBenchmarkRun, assertSingleBackend, assignHeldOutTag, attachCostToReport, attest, attributeCounterfactuals, bisect, blendHeldout, blockingKnowledgeEval, buildAgreementJudge, buildDriverSystemPrompt, buildProductBenchmarkManifest, buildReflectionPrompt, buildReviewerPrompt, buildWorkerDriverSystemPrompt, cachedJudge, canaryLeakView, canonicalJson, capabilityHeadroom, causalAttribution, checkBehavioralCanary, checkCanaries, checkSlos, checkTraceContracts, classifyTreatment, clusteredPairedBinary, codeExecutionJudge, coherenceJudge, collectionPreserved, commentsForSource, commitBisect, comparePairedArms, compareReferenceReplay, compilerJudge, computeExperimentStats, contentHash, contractJudge, costReport, createAntiSlopJudge, createCustomJudge, createDefaultReviewer, createDomainExpertJudge, createIntentMatchJudge, createSandboxPool, crossTraceDiff, dataDescriptionBits, decideNextUserTurn, decideReferenceReplayPromotion, decideReferenceReplayRunPromotion, defaultBlendWeights, defaultJudges, defaultParseStudentLabel, defaultReferenceReplayMatcher, defaultRenderStudentPrompt, deployGateLayer, diffScorecard, discoverPersonas, distillPlaybook, ensembleJudge, errorStreakDetector, estimateCost, estimateTokens, evaluateContract, evaluateOracles, evaluateTraceContract, executeScenario, expandProfileAxes, expectAgent, exportProductBenchmark, exportProductBenchmarkRuns, exportRewardModel, extractAssetUrls, extractErrorCount, fieldAgreement, fileContains, fileExists, fileExperimentStore, fileVerdictCache, findAutoMatchNoExpectation, findConstructorCwdDropped, findFallbackToPass, findLiteralTruePass, findProductBenchmarkArtifacts, findSkipCountsAsPass, flowLayer, fnv1a32, formatBenchmarkReport, formatDriverReport, formatFindings, formatScorecardDiff, gateTreatmentApplied, gateTreatmentFromMetrics, gateTreatmentFromSpans, gateTreatmentFromToolSpans, ghCliClient, gitProvenanceReader, precision as goldenPrecision, gradeOnHidden, harnessAxisOf, hashContent, hashToUnit, hiddenGrade, htmlContainsElement, httpGithubClient, improvementVerdict, inMemoryExperimentStore, inMemoryReferenceReplayStore, inMemoryRunRecordBackend, inMemoryVerdictCache, isHiddenDestination, isModelPriced, isOtelConfigured, jsonShape, jsonlReferenceReplayStore, jsonlRunRecordBackend, judgeFamily, keyPreserved, knowledgeReadinessTracePayload, leaderboard, linterJudge, loadGoldScenarios, loadScorecard, loadScorerFromGrader, localCommandRunner, lowercaseMutator, makeEvalTools, matchGoldens, matchSpan, mergeLayerResults, mergeSteeringBundle, modelDescriptionBits, multiToolchainLayer, noProgressDetector, notBlocked, observeAll, pairArms, paraphraseRobustness, paraphraseRobustnessScenarios, parseGoldJsonl, parseReflectionResponse, partitionHeldOut, passOrthogonality, pixelDeltaRatio, politenessPrefixMutator, preflightModels, printDriverSummary, productBenchmarkIntegrityFailures, productBenchmarkMutableSurfaces, productBenchmarkRepoIdentity, productBenchmarkSplits, index as profile, promptBisect, proposeSynthesisTargets, readProductBenchmarkManifest, readProductBenchmarkRecords, recordRuns, recordRunsToScorecard, referenceReplayRunsToSteeringRows, referenceReplayScenarioToRunScore, regexMatches, renderMarkdownReport, renderPlaybookMarkdown, renderSteeringText, repeatedActionDetector, replayScorerOverCorpus, replayTraceThroughJudge, resolveModelPricing, resolveSeat, routeFields, rowCount, rowWhere, runAssertions, runBehavioralCanaries, runCanaries, runCounterfactual, runDistillation, runE2EWorkflow, runExpectations, runHarnessExperiment, runIntentMatchJudge, runJudgeFleet, runKeywordCoverageJudge, runKeywordCoverageJudgeUrl, runLiveProof, runRecordToProductBenchmarkRecord, runReferenceReplay, runScore, runSelfPlay, scanForMuffledGates, scoreContinuity, scoreKnowledgeReadiness, scorePrReviewComments, scorePrReviewSource, scoreReferenceReplay, seatPresets, securityJudge, selectHarnessVariant, sentenceReorderMutator, splitGold, statusAdvanced, summarizeHarnessResults, summarizePrReviewBenchmark, testJudge, textInSnapshot, toLangfuseEnvelope, toOpenAiTool, toPrometheusText, traceContract, traceJudge, traceJudgeEnsemble, tracedAnalyzeTraces, typoMutator, urlContains, userQuestionsForKnowledgeGaps, validateProductBenchmarkManifest, validateProductBenchmarkRecord, validateProductBenchmarkRun, verifyAttestation, visualDiff, viteDeployRunner, weightedRecall, whitespaceCollapseMutator, withHeldoutBlend, withJudgeRetry, withOtelPipeline, wranglerDeployRunner };
6884
+ export { ATTESTATION_ALGORITHM, type ActiveLearningOptions, type AdapterRun, AgentDriver, type AgentDriverConfig, AgentEvalError, type AgentProfileRuntimeReceipt, type AgreementResult, type AlignmentOp, AnalyzeTracesInput, AnalyzeTracesOptions, AnalyzeTracesResult, type AntiSlopConfig, type AntiSlopIssue, type AntiSlopReport, type AssertCapabilityHeadroomOptions, type AssertCrossFamilyOptions, type AssertSingleBackendOptions, type AttestationProvenance, type AttestationVerification, type AttestedReport, type AutoPrClient, AxGepaSteeringOptimizer, type AxSteeringOptimizerConfig, type BackendDescriptor, BaselineReport, BehaviorAssertion, BehavioralMetrics, BenchmarkReport, BenchmarkRunner, BenchmarkRunnerConfig, type BisectOptions, type BisectResult, type BisectStep, type BlendWeights, BudgetBreachError, BudgetGuard, BudgetLedgerEntry, BudgetSpec, type BuildAgreementJudgeOptions, CODING_HARNESSES, type CachedJudge, type CachedJudgeOptions, CallExpectation, type CanaryAlert, type CanaryKind, type CanaryLeak, type CanaryOptions, type CanaryReport, type CanarySeverity, type CandidateComparison, type CandidateScenario, type CapabilityHeadroomOptions, type CapabilityHeadroomResult, type CausalAttributionReport, type CellVerdict, ChannelRollup, ChatRequest, CheckResult, type ClusterBootstrapInterval, type ClusterSignFlipAlternative, type ClusterSignFlipResult, type ClusteredBinaryCluster, type ClusteredMatchedPair, type ClusteredPairedBinaryOptions, type ClusteredPairedBinaryResult, type ClusteredPairedBinaryStatistics, CollectedArtifacts, type CommandRunner, type CompareLabels, CompletionCriterion, ConfigError, type ContinuityCheck, type ContinuityCheckResult, type ContinuityReport, type ContinuitySnapshotPair, type ContractCheckResult, type ContractJudgeOptions, type ContractMetric, type ContractReport, type ContractRule, type ContractRuleKind, type ContractSpan, type ContractVerdict, type ContractViolation, ControlEvalResult, ControlSeverity, ConvergenceTracker, CorrectnessChecker, type CostEntry, CostLedger, type CostReport, type CostSummary, CostTracker, type CounterfactualContext, type CounterfactualMutation, type CounterfactualResult, type CounterfactualRunner, CreateChatClientOpts, type CreateDefaultReviewerOptions, type CreateExperimentInput, type CreateSandboxPoolOpts, CrossFamilyError, type CrossTraceDiff, type CrossTraceDiffOptions, DEFAULT_AGENT_SLOS, DEFAULT_FINDERS, DEFAULT_HARNESS_OBJECTIVES, DEFAULT_MUTATION_PRIMITIVES, DEFAULT_MUTATORS, DEFAULT_PR_REVIEW_SCORE_WEIGHTS, DEFAULT_SEVERITY_WEIGHTS, type DataAcquisitionPlan, Dataset, DatasetScenario, type DecideNextUserTurnOpts, DefaultVerdict, type DeployFamily, type DeployGateLayerInput, type DeployRunResult, type DeployRunner, type DescriptionLengthCandidate, type DescriptionLengthConfig, type DescriptionLengthDecision, type DescriptionLengthEvidence, DescriptionLengthGate, type DescriptionLengthRejectionCode, type DetectorEvent, type DetectorSeverity, type DetectorSignal, type DiffScorecardOptions, type DirEntry, type DiscoverPersonasOptions, type DiscoveredPersona, DriverResult, DriverState, DualAgentBench, type DualAgentBenchConfig, type DualAgentReport, type DualAgentRound, type DualAgentScenario, type DualAgentScenarioResult, ERROR_COUNT_PATTERNS, type EnsembleAggregate, type EnsembleJudgeOptions, type ErrorCountPattern, type ErrorStreakOptions, type EvalToolDef, EvalTraceStore, type EvolutionRound, type ExecutorConfig, type Expectation, type Experiment, type ExperimentProvenance, type ExperimentRep, type ExperimentStats, type ExperimentStore, ExperimentTracker, type ExperimentTrackerOptions, type ExperimentVerdict, type ExportedRewardModel, type ExtractOptions, type ExtractResult, type FactorContribution, type FactorialCell, FailureClass, FeedbackLabel, FeedbackTrajectory, FeedbackTrajectoryStore, type FieldAgreementSpec, type FieldDestination, type FileChange, type FlowAction, type FlowLayerEnv, type FlowLayerFactoryInput, type FlowRunner, type FlowRunnerStepResult, type FlowSpec, type FlowStep, type GhCliClientOptions, type GoldScenario, type GoldSplit, type GoldenSeverity, type GoldenSpec, HARNESS_NATIVE_MODEL, type HarnessAdapter, HarnessConfig, type HarnessExperimentConfig, type HarnessExperimentResult, type HarnessIntervention, type HarnessRunRequest, type HarnessRunResult, type HarnessScenario, type HarnessSelection, type HarnessVariant, type HarnessVariantReport, type HeadroomClass, type HeadroomInput, type HeldOutPartition, type HiddenCriteriaGrader, type HiddenGradeResult, type HiddenLeak, HoldoutAuditor, type HttpGithubClientOptions, INTENT_MATCH_JUDGE_VERSION, type ImageData, type ImprovementThresholds, type ImprovementVerdictResult, InMemoryWorkspaceInspector, type InferenceScorer, type InspectorContext, type IntentMatchInput, type IntentMatchOptions, type IntentMatchResult, type InteractionContribution, JudgeError, type JudgeFamily, type JudgeFleetOptions, JudgeFn, JudgeParseError, type JudgeReplayResult, type JudgeRetryOutcome, type JudgeRetryPolicy, JudgeRunner, type JudgeScoreInput, type JudgeVerdict, type KeywordConceptSpec, type KeywordCoverageFinding, type KeywordCoverageOptions, type KeywordCoverageResult, type KnowledgeAcquisitionMode, type KnowledgeBundle, type KnowledgeFallbackPolicy, type KnowledgeFreshness, type KnowledgeImportance, type KnowledgeReadinessReport, type KnowledgeRecommendedAction, type KnowledgeRequirement, type KnowledgeRequirementCategory, type KnowledgeResponsibleSurface, type KnowledgeSensitivity, type LangfuseEnvelope, type LangfuseGeneration, type LangfuseScore, Layer, LayerResult, type LeaderboardOptions, type LeaderboardRow, type LiveProofArtifact, type LiveProofConfig, type LiveProofContext, type LiveProofResult, LlmClientOptions, LlmSpan, LockedJsonlAppender, MODEL_PRICING, type MakeEvalToolsConfig, type MatchResult, type MatcherResult, type MeasurementPolicy, type MergeOptions, MetricsCollector, type ModelCostRollup, type ModelPreflight, type ModelSeats, ModelsUnreachableError, type MuffledFinder, type MuffledFinding, type MultiToolchainLayerConfig, type Mutator, Mutex, type NoLeakOptions, type NoProgressOptions, Objective, type Oracle, type OracleObservation, type OracleReport, type OracleResult, type OrthogonalityInput, type OrthogonalityResult, OtelExportConfig, OtelExporter, type OtelPipelineHandle, type OtelPipelineOptions, PairwiseSteeringOptimizer, type ParaphraseRobustnessScenarioInput, type ParaphraseRobustnessScenarioResult, ParetoResult, type ParseStudentLabel, type PartitionHeldOutOptions, PersonaConfig, type Playbook, type PlaybookEntry, type PoolSlot, type PrReviewAuditCase, type PrReviewBenchmarkSummary, type PrReviewComment, type PrReviewMatchedFinding, type PrReviewOutcome, type PrReviewReferenceFinding, type PrReviewScore, type PrReviewScoreWeights, type PrReviewSeverity, type PrReviewSource, type PreflightModelsOptions, type PreflightOutcome, type ProductBenchmarkArm, type ProductBenchmarkArtifactPaths, type ProductBenchmarkBudgets, type ProductBenchmarkExportOptions, type ProductBenchmarkExportResult, type ProductBenchmarkManifest, type ProductBenchmarkProfileRef, type ProductBenchmarkRecord, type ProductBenchmarkRepoRef, type ProductBenchmarkRunInput, type ProductBenchmarkScenario, type ProductBenchmarkSingleRunExportOptions, type ProductBenchmarkSplit, type ProductBenchmarkSubstrateVersions, type ProductBenchmarkValidationReport, ProductClient, ProductClientConfig, type ProfileAxisSpec, type PromptHandle, PromptRegistry, type ProposeAutomatedPullRequestInput, type ProposeAutomatedPullRequestResult, type ProvenanceReader, type RecordRunsOptions, type ReferenceMatchResult, type ReferenceReplayAdapter, type ReferenceReplayAdapterFn, type ReferenceReplayAdapterLike, type ReferenceReplayAggregate, type ReferenceReplayCandidate, type ReferenceReplayCase, type ReferenceReplayCaseRun, type ReferenceReplayExecutionScenario, type ReferenceReplayItem, type ReferenceReplayMatch, type ReferenceReplayMatchStrategy, type ReferenceReplayMatcher, type ReferenceReplayPromotionDecision, type ReferenceReplayPromotionPolicy, type ReferenceReplayRun, type ReferenceReplayRunContext, type ReferenceReplayRunOptions, type ReferenceReplayRunStore, type ReferenceReplayScenario, type ReferenceReplayScenarioScore, type ReferenceReplayScore, type ReferenceReplayScoreOptions, type ReferenceReplaySplit, type ReferenceReplaySplitComparison, type ReferenceReplaySteeringRowsOptions, type ReflectionContext, type ReflectionProposal, ReleaseConfidenceScorecard, ReleaseConfidenceThresholds, type RenderStudentPrompt, type RepeatedActionOptions, type RepoRef, type ReviewerMemoryEntry, type ReviewerOutput, type ReviewerPromptInput, type ReviewerSoftFailDefaults, type ReviewerVerificationSummary, type RobustnessResult, type RoutedField, Run, type RunCommandInput, type RunCommandResult, type RunDistillationOptions, type RunDistillationResult, RunFilter, RunRecord, type RunRecordBackend, type RunRecordFilter, RunScore, RunScoreWeights, RunSplitTag, RunTrace, type RuntimeResolution, SandboxDriver, SandboxHarnessResult, type SandboxJudgeKind, type SandboxJudgeResult, type SandboxJudgeSpec, type SandboxPool, type ScanOptions, Scenario, type ScenarioCost, ScenarioFile, ScenarioRegistry, ScenarioResult, type ScoreKnowledgeReadinessOptions, type Scorecard, type ScorecardCell, type ScorecardCellDiff, type ScorecardDiff, type ScorecardEntry, type ScorecardLogLine, type ScoredTarget, type SeatName, type SeatPresetName, SeatUnsetError, type SelfPlayOptions, type SelfPlayProposer, type SelfPlayScorer, type SerializedRegex, Severity, type SingleBackendDivergence, SingleBackendError, type SingleBackendField, type SingleBackendReport, type Slo, type SloCheckResult, type SloComparator, type SloReport, type SloSeverity, type SlopCategory, type SlotFactory, Span, type SpanPredicate, type SplitGoldOptions, type SteeringBundle, type SteeringDelta, type SteeringOptimizationResult, type SteeringOptimizationRow, type SteeringOptimizationSelector, type SteeringOptimizerBackend, type SteeringOptimizerConfig, type SteeringRolePrompt, type StepAttribution, type StreamingDetector, type SynthesisReason, type SynthesisTarget, type TaskHeadroom, TestResult, type TextMatcher, type ThresholdContract, TokenCounter, type TokenSpec, type ToolMatcher, ToolSpan, TraceAnalystSpan, type TraceContract, TraceContractBuilder, TraceEmitter, TraceStore, type TracedAnalystOptions, type TracedJudgeOptions, Trajectory, TrajectoryStep, type TreatmentClass, type TreatmentGate, type TreatmentGateInput, type TreatmentGateOptions, type TrialTrace, TurnMetrics, UI_FINDING_SEVERITIES, UI_LENSES, UNIVERSAL_FINDERS, type UiFinding, type UiFindingScreenshot, type UiFindingSeverity, type UiLens, type UserQuestion, type VerdictCacheStats, type VerdictCacheStore, VerifyContext, type VisualDiffOptions, type VisualDiffResult, type ViteDeployRunnerInput, type WorkerDriverContext, type WorkflowTopology, type WorkspaceAssertion, type WorkspaceAssertionResult, type WorkspaceInspector, type WorkspaceSnapshot, type WranglerDeployRunnerInput, acquisitionPlansForKnowledgeGaps, adversarialJudge, agentProfileHash, agentProfileId, agentProfileModelId, agentVisibleFields, aggregateJudgeVerdicts, aggregatePrReviewScore, analyzeAntiSlop, appendScorecard, assertCapabilityHeadroom, assertCrossFamily, assertModelsServed, assertNoHiddenLeak, assertProductBenchmarkRun, assertSingleBackend, assignHeldOutTag, attachCostToReport, attest, attributeCounterfactuals, bisect, blendHeldout, blockingKnowledgeEval, buildAgreementJudge, buildDriverSystemPrompt, buildProductBenchmarkManifest, buildReflectionPrompt, buildReviewerPrompt, buildWorkerDriverSystemPrompt, cachedJudge, canaryLeakView, canonicalJson, capabilityHeadroom, causalAttribution, checkBehavioralCanary, checkCanaries, checkSlos, checkTraceContracts, classifyTreatment, clusteredPairedBinary, codeExecutionJudge, coherenceJudge, collectionPreserved, commentsForSource, commitBisect, compareReferenceReplay, compilerJudge, computeExperimentStats, contentHash, contractJudge, costReport, createAntiSlopJudge, createCustomJudge, createDefaultReviewer, createDomainExpertJudge, createIntentMatchJudge, createSandboxPool, crossTraceDiff, dataDescriptionBits, decideNextUserTurn, decideReferenceReplayPromotion, decideReferenceReplayRunPromotion, defaultBlendWeights, defaultJudges, defaultParseStudentLabel, defaultReferenceReplayMatcher, defaultRenderStudentPrompt, deployGateLayer, diffScorecard, discoverPersonas, distillPlaybook, ensembleJudge, errorStreakDetector, estimateCost, estimateTokens, evaluateContract, evaluateOracles, evaluateTraceContract, executeScenario, expandProfileAxes, expectAgent, exportProductBenchmark, exportProductBenchmarkRuns, exportRewardModel, extractAssetUrls, extractErrorCount, fieldAgreement, fileContains, fileExists, fileExperimentStore, fileVerdictCache, findAutoMatchNoExpectation, findConstructorCwdDropped, findFallbackToPass, findLiteralTruePass, findProductBenchmarkArtifacts, findSkipCountsAsPass, flowLayer, fnv1a32, formatBenchmarkReport, formatDriverReport, formatFindings, formatScorecardDiff, gateTreatmentApplied, gateTreatmentFromMetrics, gateTreatmentFromSpans, gateTreatmentFromToolSpans, ghCliClient, gitProvenanceReader, precision as goldenPrecision, gradeOnHidden, harnessAxisOf, hashContent, hashToUnit, hiddenGrade, htmlContainsElement, httpGithubClient, improvementVerdict, inMemoryExperimentStore, inMemoryReferenceReplayStore, inMemoryRunRecordBackend, inMemoryVerdictCache, isHiddenDestination, isModelPriced, isOtelConfigured, jsonShape, jsonlReferenceReplayStore, jsonlRunRecordBackend, judgeFamily, keyPreserved, knowledgeReadinessTracePayload, leaderboard, linterJudge, loadGoldScenarios, loadScorecard, loadScorerFromGrader, localCommandRunner, lowercaseMutator, makeEvalTools, matchGoldens, matchSpan, mergeLayerResults, mergeSteeringBundle, modelDescriptionBits, multiToolchainLayer, noProgressDetector, notBlocked, observeAll, paraphraseRobustness, paraphraseRobustnessScenarios, parseGoldJsonl, parseReflectionResponse, partitionHeldOut, passOrthogonality, pixelDeltaRatio, politenessPrefixMutator, preflightModels, printDriverSummary, productBenchmarkIntegrityFailures, productBenchmarkMutableSurfaces, productBenchmarkRepoIdentity, productBenchmarkSplits, index as profile, promptBisect, proposeSynthesisTargets, readProductBenchmarkManifest, readProductBenchmarkRecords, recordRuns, recordRunsToScorecard, referenceReplayRunsToSteeringRows, referenceReplayScenarioToRunScore, regexMatches, renderMarkdownReport, renderPlaybookMarkdown, renderSteeringText, repeatedActionDetector, replayScorerOverCorpus, replayTraceThroughJudge, resolveModelPricing, resolveSeat, routeFields, rowCount, rowWhere, runAssertions, runBehavioralCanaries, runCanaries, runCounterfactual, runDistillation, runE2EWorkflow, runExpectations, runHarnessExperiment, runIntentMatchJudge, runJudgeFleet, runKeywordCoverageJudge, runKeywordCoverageJudgeUrl, runLiveProof, runRecordToProductBenchmarkRecord, runReferenceReplay, runScore, runSelfPlay, scanForMuffledGates, scoreContinuity, scoreKnowledgeReadiness, scorePrReviewComments, scorePrReviewSource, scoreReferenceReplay, seatPresets, securityJudge, selectHarnessVariant, sentenceReorderMutator, splitGold, statusAdvanced, summarizeHarnessResults, summarizePrReviewBenchmark, testJudge, textInSnapshot, toLangfuseEnvelope, toOpenAiTool, toPrometheusText, traceContract, traceJudge, traceJudgeEnsemble, tracedAnalyzeTraces, typoMutator, urlContains, userQuestionsForKnowledgeGaps, validateProductBenchmarkManifest, validateProductBenchmarkRecord, validateProductBenchmarkRun, verifyAttestation, visualDiff, viteDeployRunner, weightedRecall, whitespaceCollapseMutator, withHeldoutBlend, withJudgeRetry, withOtelPipeline, wranglerDeployRunner };
package/dist/index.js CHANGED
@@ -47,6 +47,7 @@ import {
47
47
  agentProfileModelId,
48
48
  codeExecutionJudge,
49
49
  coherenceJudge,
50
+ comparePairedArms,
50
51
  completionVerdict,
51
52
  createCustomJudge,
52
53
  createDomainExpertJudge,
@@ -57,9 +58,10 @@ import {
57
58
  extractProducedState,
58
59
  harnessAxisOf,
59
60
  llmJudge,
61
+ pairArms,
60
62
  parseCorrectnessResponse,
61
63
  verifyCompletion
62
- } from "./chunk-VHPD6AXX.js";
64
+ } from "./chunk-KDCMEZDI.js";
63
65
  import {
64
66
  DEFAULT_MUTATION_PRIMITIVES,
65
67
  DEFAULT_RED_TEAM_CORPUS,
@@ -107,7 +109,6 @@ import {
107
109
  DEFAULT_COMPLEXITY_WEIGHTS,
108
110
  FindingsStore,
109
111
  LockedJsonlAppender,
110
- Mutex,
111
112
  RunCritic,
112
113
  SEMANTIC_CONCEPT_JUDGE_VERSION,
113
114
  SKILL_USAGE_ANALYST,
@@ -118,13 +119,14 @@ import {
118
119
  defaultIsMaterial,
119
120
  diffFindings,
120
121
  runSemanticConceptJudge
121
- } from "./chunk-4FRTXH2M.js";
122
+ } from "./chunk-I2HNIE6N.js";
122
123
  import {
123
124
  buildDefaultAnalystRegistry,
124
125
  computeTraceMetrics
125
126
  } from "./chunk-LVTGFSHF.js";
126
127
  import {
127
128
  DEFAULT_RUN_SCORE_WEIGHTS,
129
+ Mutex,
128
130
  POLICY_EDIT_AXES,
129
131
  POLICY_EDIT_TARGET_SURFACES,
130
132
  PolicyEditValidationError,
@@ -139,7 +141,7 @@ import {
139
141
  policyEditsFromFindings,
140
142
  scorePolicyEditReadiness,
141
143
  validatePolicyEdit
142
- } from "./chunk-HJWNHCD5.js";
144
+ } from "./chunk-AN5UYSVD.js";
143
145
  import {
144
146
  AnalystRegistry,
145
147
  DEFAULT_TRACE_ANALYST_KINDS,
@@ -1302,155 +1304,6 @@ async function runE2EWorkflow(client, name, workflow) {
1302
1304
  };
1303
1305
  }
1304
1306
 
1305
- // src/paired-arms.ts
1306
- function pairArms(rows, opts) {
1307
- const { baselineArm, treatmentArm } = opts;
1308
- if (baselineArm === treatmentArm) {
1309
- throw new ValidationError(
1310
- `pairArms: baselineArm and treatmentArm are both '${baselineArm}' \u2014 an arm cannot be compared to itself`
1311
- );
1312
- }
1313
- const byArm = /* @__PURE__ */ new Map();
1314
- const armsSeen = /* @__PURE__ */ new Set();
1315
- for (const row of rows) {
1316
- armsSeen.add(row.arm);
1317
- if (row.arm !== baselineArm && row.arm !== treatmentArm) continue;
1318
- const byKey = byArm.get(row.arm) ?? /* @__PURE__ */ new Map();
1319
- const group = byKey.get(row.pairKey) ?? [];
1320
- group.push(row);
1321
- byKey.set(row.pairKey, group);
1322
- byArm.set(row.arm, byKey);
1323
- }
1324
- for (const arm of [baselineArm, treatmentArm]) {
1325
- if (!byArm.has(arm)) {
1326
- const seen = [...armsSeen].sort().join(", ") || "<none>";
1327
- throw new ValidationError(`pairArms: no rows for arm '${arm}' (arms present: ${seen})`);
1328
- }
1329
- }
1330
- const baselineByKey = byArm.get(baselineArm);
1331
- const treatmentByKey = byArm.get(treatmentArm);
1332
- const allKeys = [.../* @__PURE__ */ new Set([...baselineByKey.keys(), ...treatmentByKey.keys()])].sort();
1333
- const pairs = [];
1334
- const unpairedBaseline = [];
1335
- const unpairedTreatment = [];
1336
- for (const pairKey2 of allKeys) {
1337
- const b = baselineByKey.get(pairKey2) ?? [];
1338
- const t = treatmentByKey.get(pairKey2) ?? [];
1339
- if (b.length <= 1 && t.length <= 1) {
1340
- if (b.length === 1 && t.length === 1) {
1341
- pairs.push({ pairKey: pairKey2, repIndex: 0, baseline: b[0], treatment: t[0] });
1342
- } else {
1343
- unpairedBaseline.push(...b);
1344
- unpairedTreatment.push(...t);
1345
- }
1346
- continue;
1347
- }
1348
- const bByRep = indexByRepKey(b, pairKey2, baselineArm);
1349
- const tByRep = indexByRepKey(t, pairKey2, treatmentArm);
1350
- const repKeys = [.../* @__PURE__ */ new Set([...bByRep.keys(), ...tByRep.keys()])].sort();
1351
- let repIndex = 0;
1352
- for (const repKey of repKeys) {
1353
- const baseline = bByRep.get(repKey);
1354
- const treatment = tByRep.get(repKey);
1355
- if (baseline !== void 0 && treatment !== void 0) {
1356
- pairs.push({ pairKey: pairKey2, repIndex: repIndex++, baseline, treatment });
1357
- } else if (baseline !== void 0) {
1358
- unpairedBaseline.push(baseline);
1359
- } else if (treatment !== void 0) {
1360
- unpairedTreatment.push(treatment);
1361
- }
1362
- }
1363
- }
1364
- return { pairs, unpairedBaseline, unpairedTreatment };
1365
- }
1366
- function indexByRepKey(group, pairKey2, arm) {
1367
- const byRep = /* @__PURE__ */ new Map();
1368
- for (const row of group) {
1369
- if (row.repKey === void 0) {
1370
- throw new ValidationError(
1371
- `pairArms: pairKey '${pairKey2}' has multiple reps in an arm, but a row in arm '${arm}' is missing repKey \u2014 multi-rep items require an explicit repKey on every row so reps pair by identity (pairing reps by outcome or by index would bias the paired statistics)`
1372
- );
1373
- }
1374
- if (byRep.has(row.repKey)) {
1375
- throw new ValidationError(
1376
- `pairArms: duplicate repKey '${row.repKey}' for pairKey '${pairKey2}' in arm '${arm}' \u2014 (pairKey, repKey) must uniquely identify a rep within an arm`
1377
- );
1378
- }
1379
- byRep.set(row.repKey, row);
1380
- }
1381
- return byRep;
1382
- }
1383
- function comparePairedArms(rows, opts) {
1384
- const { pairs, unpairedBaseline, unpairedTreatment } = pairArms(rows, opts);
1385
- let correctness = null;
1386
- const baselinePass = [];
1387
- const treatmentPass = [];
1388
- for (const pair of pairs) {
1389
- if (pair.baseline.pass === void 0 || pair.treatment.pass === void 0) continue;
1390
- baselinePass.push(pair.baseline.pass ? 1 : 0);
1391
- treatmentPass.push(pair.treatment.pass ? 1 : 0);
1392
- }
1393
- if (baselinePass.length > 0) {
1394
- const mc = mcnemar(baselinePass, treatmentPass);
1395
- correctness = {
1396
- b10: mc.b,
1397
- b01: mc.c,
1398
- mcnemar: mc,
1399
- riskDifference: pairedRiskDifference(baselinePass, treatmentPass)
1400
- };
1401
- }
1402
- const metricNames = opts.metricNames ?? [
1403
- ...new Set(
1404
- pairs.flatMap((p) => [
1405
- ...Object.keys(p.baseline.metrics ?? {}),
1406
- ...Object.keys(p.treatment.metrics ?? {})
1407
- ])
1408
- )
1409
- ].sort();
1410
- const metricDeltas = metricNames.map((name) => {
1411
- const before = [];
1412
- const after = [];
1413
- let nMissing = 0;
1414
- for (const pair of pairs) {
1415
- const b = metricValue(pair.baseline, name);
1416
- const t = metricValue(pair.treatment, name);
1417
- if (b === void 0 || t === void 0) {
1418
- nMissing++;
1419
- continue;
1420
- }
1421
- before.push(b);
1422
- after.push(t);
1423
- }
1424
- const bootstrapCi2 = before.length === 0 ? null : pairedBootstrap(before, after, opts.bootstrap);
1425
- return {
1426
- name,
1427
- n: before.length,
1428
- nMissing,
1429
- medianDelta: bootstrapCi2 === null ? Number.NaN : bootstrapCi2.median,
1430
- meanDelta: bootstrapCi2 === null ? Number.NaN : bootstrapCi2.mean,
1431
- bootstrapCi: bootstrapCi2,
1432
- wilcoxon: before.length === 0 ? null : wilcoxonSignedRank(before, after)
1433
- };
1434
- });
1435
- return {
1436
- nPairs: pairs.length,
1437
- nUnpairedBaseline: unpairedBaseline.length,
1438
- nUnpairedTreatment: unpairedTreatment.length,
1439
- correctness,
1440
- metricDeltas
1441
- };
1442
- }
1443
- function metricValue(row, name) {
1444
- const v = row.metrics?.[name];
1445
- if (v === void 0) return void 0;
1446
- if (!Number.isFinite(v)) {
1447
- throw new ValidationError(
1448
- `comparePairedArms: non-finite value for metric '${name}' on pairKey '${row.pairKey}' (arm '${row.arm}'): ${v}`
1449
- );
1450
- }
1451
- return v;
1452
- }
1453
-
1454
1307
  // src/clustered-paired-binary.ts
1455
1308
  var DEFAULT_BOOTSTRAP_RESAMPLES = 1e4;
1456
1309
  var DEFAULT_SIGN_FLIP_RESAMPLES = 1e5;