@tangle-network/agent-eval 0.135.1 → 0.135.2
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +17 -0
- package/dist/{analyze-runs-DMo3Lb_y.d.ts → analyze-runs-Cda5Xkj1.d.ts} +3 -3
- package/dist/{analyze-runs-DMo3Lb_y.d.ts.map → analyze-runs-Cda5Xkj1.d.ts.map} +1 -1
- package/dist/{analyze-runs-qk8op0tN.js → analyze-runs-jjCmF8pU.js} +14 -7
- package/dist/analyze-runs-jjCmF8pU.js.map +1 -0
- package/dist/{baseline-BaPxoROc.js → baseline-BUeFcgrn.js} +2 -2
- package/dist/{baseline-BaPxoROc.js.map → baseline-BUeFcgrn.js.map} +1 -1
- package/dist/benchmarks/index.d.ts +1 -1
- package/dist/benchmarks/index.js +1 -1
- package/dist/{benchmarks-DwFrCrCS.js → benchmarks-Mtu251Jz.js} +3 -3
- package/dist/{benchmarks-DwFrCrCS.js.map → benchmarks-Mtu251Jz.js.map} +1 -1
- package/dist/builder-eval/index.js +1 -1
- package/dist/campaign/index.d.ts +2 -2
- package/dist/campaign/index.js +2 -2
- package/dist/{campaign-B0oyngTs.js → campaign-RVIqtJh0.js} +5 -5
- package/dist/{campaign-B0oyngTs.js.map → campaign-RVIqtJh0.js.map} +1 -1
- package/dist/{client-BIyh1RCr.d.ts → client-DcvgkaZi.d.ts} +13 -4
- package/dist/client-DcvgkaZi.d.ts.map +1 -0
- package/dist/contract/index.d.ts +3 -3
- package/dist/contract/index.d.ts.map +1 -1
- package/dist/contract/index.js +17 -5
- package/dist/contract/index.js.map +1 -1
- package/dist/{eval-campaign-aMG73hvX.js → eval-campaign-Cc8WZJ6b.js} +2 -2
- package/dist/{eval-campaign-aMG73hvX.js.map → eval-campaign-Cc8WZJ6b.js.map} +1 -1
- package/dist/hosted/index.d.ts +1 -1
- package/dist/{index-BoJNQR6n.d.ts → index-B4Fjfo5U.d.ts} +93 -15
- package/dist/index-B4Fjfo5U.d.ts.map +1 -0
- package/dist/{index-C21xKtxu.d.ts → index-CQsJcqch.d.ts} +3 -3
- package/dist/{index-C21xKtxu.d.ts.map → index-CQsJcqch.d.ts.map} +1 -1
- package/dist/{index-DSC51roc2.d.ts → index-DSC51roc.d.ts} +1 -1
- package/dist/index-DSC51roc.d.ts.map +1 -0
- package/dist/index.d.ts +41 -10
- package/dist/index.d.ts.map +1 -1
- package/dist/index.js +131 -53
- package/dist/index.js.map +1 -1
- package/dist/matrix/index.d.ts +1 -1
- package/dist/meta-eval/index.d.ts +1 -2
- package/dist/meta-eval/index.d.ts.map +1 -1
- package/dist/meta-eval/index.js +2 -2
- package/dist/multishot/index.d.ts +1 -1
- package/dist/openapi.json +1 -1
- package/dist/{paired-arms-CA_8pN01.js → paired-arms-BbFKrAU-.js} +2 -2
- package/dist/{paired-arms-CA_8pN01.js.map → paired-arms-BbFKrAU-.js.map} +1 -1
- package/dist/pipelines/index.js +2 -2
- package/dist/{release-report-BVZBmRZp.js → release-report-DooPguBc.js} +4 -3
- package/dist/{release-report-BVZBmRZp.js.map → release-report-DooPguBc.js.map} +1 -1
- package/dist/{release-report-CuULWKyk.d.ts → release-report-DpBxGGI1.d.ts} +2 -2
- package/dist/{release-report-CuULWKyk.d.ts.map → release-report-DpBxGGI1.d.ts.map} +1 -1
- package/dist/reporting.d.ts +3 -3
- package/dist/reporting.js +4 -4
- package/dist/{researcher-DVtruQ9U.d.ts → researcher-Doo95b50.d.ts} +2 -2
- package/dist/{researcher-DVtruQ9U.d.ts.map → researcher-Doo95b50.d.ts.map} +1 -1
- package/dist/{reward-hacking-DCdRK9TY.js → reward-hacking-a-kYs0-i.js} +2 -2
- package/dist/{reward-hacking-DCdRK9TY.js.map → reward-hacking-a-kYs0-i.js.map} +1 -1
- package/dist/rl.d.ts +45 -3
- package/dist/rl.d.ts.map +1 -1
- package/dist/rl.js +108 -22
- package/dist/rl.js.map +1 -1
- package/dist/{rubric-predictive-validity-D6Q6n9oq.js → rubric-predictive-validity-BJf-8ejY.js} +2 -2
- package/dist/{rubric-predictive-validity-D6Q6n9oq.js.map → rubric-predictive-validity-BJf-8ejY.js.map} +1 -1
- package/dist/{skillopt-optimization-method-C9yGg-4C.js → skillopt-optimization-method-0UmPD6aP.js} +340 -54
- package/dist/skillopt-optimization-method-0UmPD6aP.js.map +1 -0
- package/dist/{skillopt-optimization-method-DJ3l4w8W.d.ts → skillopt-optimization-method-CwSYkv35.d.ts} +39 -9
- package/dist/skillopt-optimization-method-CwSYkv35.d.ts.map +1 -0
- package/dist/{statistics-D_4Snl-5.d.ts → statistics-CKOqre5S.d.ts} +329 -3
- package/dist/statistics-CKOqre5S.d.ts.map +1 -0
- package/dist/{statistics-RwRNu2__.js → statistics-CnGCLLqc.js} +315 -2
- package/dist/statistics-CnGCLLqc.js.map +1 -0
- package/dist/{summary-report-BxtossFi.js → summary-report-BEk8OFLs.js} +11 -6
- package/dist/summary-report-BEk8OFLs.js.map +1 -0
- package/dist/{summary-report-DGp0-_XO.d.ts → summary-report-CPMINBqs.d.ts} +182 -7
- package/dist/summary-report-CPMINBqs.d.ts.map +1 -0
- package/package.json +1 -1
- package/dist/analyze-runs-qk8op0tN.js.map +0 -1
- package/dist/client-BIyh1RCr.d.ts.map +0 -1
- package/dist/index-BoJNQR6n.d.ts.map +0 -1
- package/dist/index-DSC51roc2.d.ts.map +0 -1
- package/dist/judge-calibration-DFtEMlde.d.ts +0 -146
- package/dist/judge-calibration-DFtEMlde.d.ts.map +0 -1
- package/dist/skillopt-optimization-method-C9yGg-4C.js.map +0 -1
- package/dist/skillopt-optimization-method-DJ3l4w8W.d.ts.map +0 -1
- package/dist/statistics-D_4Snl-5.d.ts.map +0 -1
- package/dist/statistics-RwRNu2__.js.map +0 -1
- package/dist/summary-report-BxtossFi.js.map +0 -1
- package/dist/summary-report-DGp0-_XO.d.ts.map +0 -1
package/dist/index.js
CHANGED
|
@@ -12,14 +12,14 @@ import { S as traceSpanKindToOpenInferenceKind, a as SpanNotFoundError, b as cla
|
|
|
12
12
|
import { a as clamp01, c as makeFinding, i as aggregateRunScore, l as makeProposalFinding, r as DEFAULT_RUN_SCORE_WEIGHTS, s as computeFindingId } from "./proposal-findings-DCawte-y.js";
|
|
13
13
|
import { n as mapConcurrent, t as Mutex } from "./concurrency-MUjT7VjM.js";
|
|
14
14
|
import { a as RunCritic, d as defaultIsMaterial, f as diffFindings, i as runSemanticConceptJudge, n as SEMANTIC_CONCEPT_JUDGE_VERSION, o as SKILL_USAGE_ANALYST, p as LockedJsonlAppender, r as createSemanticConceptJudge, s as SkillUsageAnalyst, t as DEFAULT_COMPLEXITY_WEIGHTS, u as FindingsStore } from "./semantic-concept-judge-Btozx3Vc.js";
|
|
15
|
-
import { At as dominates, B as surfaceContentHash, Bt as
|
|
16
|
-
import { A as
|
|
17
|
-
import { n as pairArms, r as pairRunRecords, t as comparePairedArms } from "./paired-arms-
|
|
15
|
+
import { At as dominates, B as surfaceContentHash, Bt as assertRealAgentReceipts, Ct as llmJudge, D as runCanaries, Dt as fileVerdictCache, Et as contentHash, Ft as decidePairedPromotion, G as DEFAULT_MUTATION_PRIMITIVES, Ht as summarizeAgentReceiptIntegrity, It as pairedDecisionShape, K as buildReflectionPrompt, Lt as minimumPairsForPairedDeltaTest, Mt as paretoFrontierWithCrowding, Nt as scalarScore, Ot as inMemoryVerdictCache, Rt as pairedDeltaTest, St as hashScenarios, Tt as canonicalJson, Ut as summarizeBackendIntegrity, Vt as assertRealBackend, Wt as JudgeParseError, _t as redTeamReport, bt as Dataset, dt as REFERENCE_EQUIVALENCE_INPUT_LIMITS, ft as REFERENCE_EQUIVALENCE_JUDGE_VERSION, gt as redTeamDataset, ht as DEFAULT_RED_TEAM_CORPUS, jt as paretoFrontier, kt as crowdingDistance, mt as runReferenceEquivalenceJudge, pt as createReferenceEquivalenceJudge, q as parseReflectionResponse, vt as scoreRedTeamOutput, wt as cachedJudge, xt as HoldoutLockedError, yt as toolNamesForRun, zt as BackendIntegrityError } from "./skillopt-optimization-method-0UmPD6aP.js";
|
|
16
|
+
import { $ as selfPreference, A as pairedRiskDifference, B as requiredSampleSize, C as mulberry32, D as pairedCohensDz, E as pairedBootstrap, F as partialCredit, G as wilson, H as weightedComposite, I as passAtK, J as normalCdf, K as studentTCdf, L as pearsonR, M as pairedRiskDifferenceScore, N as pairedSignTest, O as pairedDeltaTieFraction, P as pairedTTest, Q as positionalBias, R as ranks, S as mcnemarRequiredN, T as pairedBinaryScale, U as weightedMean, V as spearmanR, W as wilcoxonSignedRank, X as calibrateJudgeContinuous, Y as calibrateJudge, Z as continuousAgreement, _ as interpretCliffs, a as MANN_WHITNEY_EXACT_MAX_WORK, b as mcnemar, c as bonferroni, d as confidenceInterval, et as verbosityBias, f as corpusInterRaterAgreement, g as interRaterReliability, h as holm, i as MANN_WHITNEY_EXACT_MAX_STATES, j as pairedRiskDifferenceExact, k as pairedMde, l as cliffsDelta, m as eProcess, n as DECISION_PAIRED_DELTA_STATISTIC, o as WILCOXON_EXACT_MAX_N, p as corpusInterRaterAgreementFromJudgeScores, r as DEFAULT_PERMUTATIONS, s as benjaminiHochberg, t as BOOTSTRAP_GATE_MIN_N, u as cohensD, v as isBinaryOutcomeVector, w as normalizeScores, x as mcnemarPower, y as mannWhitneyU, z as requiredPairedSampleSize } from "./statistics-CnGCLLqc.js";
|
|
17
|
+
import { n as pairArms, r as pairRunRecords, t as comparePairedArms } from "./paired-arms-BbFKrAU-.js";
|
|
18
18
|
import { n as llmSpanFromProvider, t as TraceEmitter } from "./emitter-CPBAhxum.js";
|
|
19
19
|
import { a as scoreOrigin, n as observedScore, o as trainingReward, r as observedSplitScore, s as trainingScore, t as isRealnessGated } from "./reward-nw2xZGZG.js";
|
|
20
20
|
import { a as isRetrievalSpan, i as isLlmSpan, n as TRACE_SCHEMA_VERSION, o as isSandboxSpan, r as isJudgeSpan, s as isToolSpan, t as FAILURE_CLASSES } from "./schema-CRhEY1SO.js";
|
|
21
21
|
import { a as roundTripRunRecord, i as parseRunRecordSafe, n as isRunRecord, o as runTaskScore, r as modelHasSnapshot, s as validateRunRecord, t as RunRecordValidationError } from "./run-record-BIwU2wdV.js";
|
|
22
|
-
import { a as evaluateReleaseConfidence, i as assertReleaseConfidence, n as bootstrapCi, r as judgeReplayGate, t as renderReleaseReport } from "./release-report-
|
|
22
|
+
import { a as evaluateReleaseConfidence, i as assertReleaseConfidence, n as bootstrapCi, r as judgeReplayGate, t as renderReleaseReport } from "./release-report-DooPguBc.js";
|
|
23
23
|
import { c as assertMintedLines, f as isRolloutLine, i as ROLLOUT_SCHEMA, m as validateRolloutLine, p as isTrainableSplit, s as assertMinted, u as assertRolloutLine } from "./schema-C6DW4ZHR.js";
|
|
24
24
|
import { i as toRewardRows, r as toJsonl, s as toSftRows } from "./exporters-q9iL-2Jf.js";
|
|
25
25
|
import { a as toHarborTrajectories, i as relabelImportedSplit, n as HARBOR_IMPORT_GAP, o as toHarborTrajectory, r as fromHarborTrajectory, t as ATIF_SCHEMA_VERSION } from "./rollout-DLSUIWLu.js";
|
|
@@ -29,18 +29,18 @@ import { D as showMeasured, E as isUnavailable, S as rollupSupervisorRuns, T as
|
|
|
29
29
|
import { n as TRACE_ANALYST_ACTOR_DESCRIPTION, r as TRACE_ANALYST_ACTOR_DESCRIPTION_VERSION, t as analyzeTraces } from "./analyst-LsnNpSkm.js";
|
|
30
30
|
import { C as planTraceInsightQuestions, E as traceAnalystOnRunComplete, S as inferDomainKeywords, T as tokenizeDomainWords, _ as buildTraceInsightContext, a as convertTraceStoresToOtlp, b as describeTraceInsightScope, c as otelRunCompleteHook, d as captureFetchToRawSink, f as otlpRowsToRunRecords, g as flattenOtlpExportToNdjson, h as otlpToTraceRunRecords, i as iterateRawCalls, l as OTEL_AGENT_EVAL_SCOPE, m as otlpToRunRecords, n as ReplayCacheMissError, o as createOtelExporter, p as otlpRowsToTraceRunRecords, r as createReplayFetch, s as createOtelTracingStore, t as ReplayCache, u as exportRunAsOtlp, v as buildTraceInsightPrompt, w as scoreTraceInsightReadiness, x as domainEvidencePattern, y as defaultTraceInsightPanel } from "./replay-CJfGLdx4.js";
|
|
31
31
|
import { n as extractUsageFromResponse, r as extractUsageFromSse, t as extractUsage } from "./extract-usage-2j25whHw.js";
|
|
32
|
-
import { B as harnessAxisOf, F as HARNESS_NATIVE_MODEL, G as parseCorrectnessResponse, H as completionVerdict, I as agentProfileHash, K as verifyCompletion, L as agentProfileId, P as CODING_HARNESSES, R as agentProfileModelId, U as createLlmCorrectnessChecker, V as extractProducedState, W as createTokenRecallChecker, z as expandProfileAxes } from "./campaign-
|
|
33
|
-
import { n as iqr, r as welchsTTest, t as compareToBaseline } from "./baseline-
|
|
32
|
+
import { B as harnessAxisOf, F as HARNESS_NATIVE_MODEL, G as parseCorrectnessResponse, H as completionVerdict, I as agentProfileHash, K as verifyCompletion, L as agentProfileId, P as CODING_HARNESSES, R as agentProfileModelId, U as createLlmCorrectnessChecker, V as extractProducedState, W as createTokenRecallChecker, z as expandProfileAxes } from "./campaign-RVIqtJh0.js";
|
|
33
|
+
import { n as iqr, r as welchsTTest, t as compareToBaseline } from "./baseline-BUeFcgrn.js";
|
|
34
34
|
import { a as judgeSpans, c as runsForScenario, i as hasCapturedToolArgs, l as toolSpans, n as argHash, o as llmSpans, r as groupBy, s as runFailureClass, t as aggregateLlm } from "./query-Di7eEQ79.js";
|
|
35
|
-
import { a as checkBehavioralCanary, i as canaryLeakView, o as checkCanaries, r as HoldoutAuditor, s as runBehavioralCanaries, t as analyzeRuns } from "./analyze-runs-
|
|
36
|
-
import { a as summaryTable, i as researchReport, n as gainHistogram, r as paretoChart, t as RESEARCH_REPORT_HARD_PAIR_FLOOR } from "./summary-report-
|
|
35
|
+
import { a as checkBehavioralCanary, i as canaryLeakView, o as checkCanaries, r as HoldoutAuditor, s as runBehavioralCanaries, t as analyzeRuns } from "./analyze-runs-jjCmF8pU.js";
|
|
36
|
+
import { a as summaryTable, i as researchReport, n as gainHistogram, r as paretoChart, t as RESEARCH_REPORT_HARD_PAIR_FLOOR } from "./summary-report-BEk8OFLs.js";
|
|
37
37
|
import { a as composeParsers, c as vitestTestParser, i as SubprocessSandboxDriver, n as DockerSandboxDriver, o as jestTestParser, r as SandboxHarness, s as pytestTestParser, t as runTestGradedScenario } from "./test-graded-scenario-BsqWLmPt.js";
|
|
38
38
|
import { a as InMemoryTraceStore, i as FileSystemTraceStore, n as assertRunCaptured, r as throwIfRunIncomplete, t as RunIntegrityError } from "./integrity-BzRbCHzi.js";
|
|
39
39
|
import { i as redactValue, n as REDACTION_VERSION, r as redactString, t as DEFAULT_REDACTION_RULES } from "./redact-7Aq1ukl-.js";
|
|
40
40
|
import { n as DEFAULT_RULES, r as classifyFailure, t as computeToolUseMetrics } from "./tool-use-metrics-DEGMKycK.js";
|
|
41
41
|
import { t as analyzeSeries } from "./series-convergence-CjO2QdRW.js";
|
|
42
|
-
import { _ as deterministicSplit, g as BENCHMARK_SPLIT_SEED, t as benchmarks_exports } from "./benchmarks-
|
|
43
|
-
import { t as runEvalCampaign } from "./eval-campaign-
|
|
42
|
+
import { _ as deterministicSplit, g as BENCHMARK_SPLIT_SEED, t as benchmarks_exports } from "./benchmarks-Mtu251Jz.js";
|
|
43
|
+
import { t as runEvalCampaign } from "./eval-campaign-Cc8WZJ6b.js";
|
|
44
44
|
import { n as pairedEvalueSequence, t as evaluateInterimReleaseConfidence } from "./sequential-Br0mAPHA.js";
|
|
45
45
|
import { AxGEPA, AxSignature, ax } from "@ax-llm/ax";
|
|
46
46
|
import { accessSync, appendFileSync, constants, cpSync, existsSync, mkdirSync, promises, readFileSync, readdirSync, statSync, writeFileSync } from "node:fs";
|
|
@@ -10170,9 +10170,10 @@ function precision(goldens, candidates, options = {}) {
|
|
|
10170
10170
|
* 1. **Productive runs**: the candidate has at least
|
|
10171
10171
|
* `minProductiveRuns` paired observations on items where BOTH
|
|
10172
10172
|
* candidate and baseline produced a real (non-silent) score.
|
|
10173
|
-
* 2. **Paired delta**: the lower bound of the
|
|
10174
|
-
*
|
|
10175
|
-
*
|
|
10173
|
+
* 2. **Paired delta**: the lower bound of the CI on the per-item
|
|
10174
|
+
* holdout delta (candidate − baseline) is strictly greater than
|
|
10175
|
+
* `pairedDeltaThreshold`. WHICH paired statistic carries that CI
|
|
10176
|
+
* depends on the outcome's shape — see "Choosing the statistic".
|
|
10176
10177
|
* 3. **Overfit gap**: the candidate's gap between search-split
|
|
10177
10178
|
* score and holdout-split score is no worse (more positive)
|
|
10178
10179
|
* than the baseline's gap by more than `overfitGapThreshold`.
|
|
@@ -10209,13 +10210,69 @@ function precision(goldens, candidates, options = {}) {
|
|
|
10209
10210
|
* "expected" scenarios to keep in sync and nothing to trust — an item counts as
|
|
10210
10211
|
* dealt because a row for it exists on at least one arm.
|
|
10211
10212
|
*
|
|
10213
|
+
* ## Choosing the statistic
|
|
10214
|
+
*
|
|
10215
|
+
* The median paired delta is the robust default, but it goes BLIND wherever
|
|
10216
|
+
* ties dominate: with more than half the pairs tied its sample value is exactly
|
|
10217
|
+
* 0 whatever the shift on the rest, and the whole bootstrap CI collapses to
|
|
10218
|
+
* [0, 0]. `low > threshold` is then "no" forever at a non-negative threshold
|
|
10219
|
+
* and "yes" forever at a negative one — the gate refuses real gains AND
|
|
10220
|
+
* promotes real regressions. Measured: 76 held-out items with 15 candidate
|
|
10221
|
+
* wins, 5 candidate losses and 56 ties is a real +13.2pp lift whose median
|
|
10222
|
+
* paired delta is 0; the median-CI gate refused it at every non-negative
|
|
10223
|
+
* threshold while PROMOTING the mirror-image −13.2pp regression at any
|
|
10224
|
+
* negative one.
|
|
10225
|
+
*
|
|
10226
|
+
* So the statistic is picked by the shape of the data, in three cases:
|
|
10227
|
+
*
|
|
10228
|
+
* 1. **Two-point (pass/fail) outcomes on ANY encoding** — every value on both
|
|
10229
|
+
* arms is either 0 or one common positive level, so {0,1} and {0,100} and
|
|
10230
|
+
* {0,5} all qualify (see `pairedBinaryScale`). The estimand is the change in
|
|
10231
|
+
* success rate, rescaled into the caller's native units. Two statistics
|
|
10232
|
+
* answer about it and BOTH must agree before anything is promoted:
|
|
10233
|
+
* `pairedRiskDifferenceScore` (Tango's score interval) supplies the
|
|
10234
|
+
* interval, and McNemar's exact test — carried on
|
|
10235
|
+
* `pairedRiskDifferenceExact` — holds a veto at any non-negative threshold.
|
|
10236
|
+
* The score interval is the one that decides because it is valid at a
|
|
10237
|
+
* NONZERO margin, which is the only thing a noninferiority threshold asks
|
|
10238
|
+
* about; the exact test's own Clopper-Pearson interval is conditional on the
|
|
10239
|
+
* discordant count and is therefore a valid interval only at margin 0.
|
|
10240
|
+
* McNemar's b/c/p ship as evidence.
|
|
10241
|
+
* 2. **Everything else** — the bootstrap CI of the MEAN paired delta. The mean
|
|
10242
|
+
* is the estimator that answers the gate's own question ("by how much did
|
|
10243
|
+
* the score move") in the caller's units, and it is well defined on every
|
|
10244
|
+
* shape the median loses: tie-dominated vectors, low-cardinality lattices
|
|
10245
|
+
* (block scores in {⅔, 1} from averaging pass/fail leaves; integer 0-100
|
|
10246
|
+
* judge dimensions), and mixed arms. Measured: 26 blocks of 3 pass/fail
|
|
10247
|
+
* leaves carrying a real +12.8pp lift, only 23% of pairs tied, give a
|
|
10248
|
+
* median CI of [0, 0.333] — a lower bound of exactly 0 refuses a real lift
|
|
10249
|
+
* at threshold 0, which is the original bug one lattice over. There is
|
|
10250
|
+
* therefore no tie-fraction cutoff: any cutoff leaves the lattice case open
|
|
10251
|
+
* on the far side of it. `heldoutSignificance` has defaulted to the mean
|
|
10252
|
+
* since #316 for the same reason.
|
|
10253
|
+
*
|
|
10254
|
+
* Set `deltaStatistic: 'median'` to force the median on every input — the
|
|
10255
|
+
* pre-0.134 behaviour, kept for callers who want outlier robustness and accept
|
|
10256
|
+
* the blindness that comes with it.
|
|
10257
|
+
*
|
|
10258
|
+
* ## Fail closed
|
|
10259
|
+
*
|
|
10260
|
+
* A ZERO-WIDTH CI carries no information about how far the estimate could be
|
|
10261
|
+
* wrong, wherever it sits. At [0, 0] it cannot tell a gain from a regression
|
|
10262
|
+
* and it clears every negative threshold. Away from zero it fails the opposite
|
|
10263
|
+
* way: n identical positive deltas give [g, g], which clears threshold 0 on no
|
|
10264
|
+
* spread at all. Both are an absence of evidence, so the gate refuses with
|
|
10265
|
+
* `indeterminate_delta` rather than answering "promote".
|
|
10266
|
+
*
|
|
10212
10267
|
* The decision carries a machine-readable `rejectionCode` plus an
|
|
10213
10268
|
* `evidence` block with every number the gate looked at, so the
|
|
10214
10269
|
* downstream researcher / paper / dashboard can re-derive the
|
|
10215
10270
|
* verdict without re-running.
|
|
10216
10271
|
*
|
|
10217
10272
|
* See also:
|
|
10218
|
-
* - `src/statistics.ts` for `pairedBootstrap` + `wilcoxonSignedRank
|
|
10273
|
+
* - `src/statistics.ts` for `pairedBootstrap` + `wilcoxonSignedRank`, and
|
|
10274
|
+
* `pairedRiskDifferenceScore` / `pairedRiskDifferenceExact` (+
|
|
10275
|
+
* `pairedBinaryScale`) for the binary path
|
|
10219
10276
|
* - `src/run-record.ts` for the input row schema
|
|
10220
10277
|
* - `src/reference-replay.ts` for the older, reference-replay-
|
|
10221
10278
|
* specific promotion path (still useful for replay-style evals).
|
|
@@ -10235,6 +10292,7 @@ var HeldOutGate = class {
|
|
|
10235
10292
|
seed;
|
|
10236
10293
|
costPerTaskCeiling;
|
|
10237
10294
|
minCoverage;
|
|
10295
|
+
deltaStatistic;
|
|
10238
10296
|
constructor(config) {
|
|
10239
10297
|
if (!config.baselineKey) throw new Error("HeldOutGate: baselineKey is required");
|
|
10240
10298
|
if (config.minCoverage !== void 0 && !(Number.isFinite(config.minCoverage) && config.minCoverage >= 0 && config.minCoverage <= 1)) throw new Error(`HeldOutGate: minCoverage must be a finite fraction in [0, 1], got ${config.minCoverage}`);
|
|
@@ -10246,6 +10304,7 @@ var HeldOutGate = class {
|
|
|
10246
10304
|
this.baselineKey = config.baselineKey;
|
|
10247
10305
|
this.resamples = config.bootstrapResamples ?? 2e3;
|
|
10248
10306
|
this.seed = config.seed;
|
|
10307
|
+
this.deltaStatistic = config.deltaStatistic ?? "auto";
|
|
10249
10308
|
if (config.costPerTaskCeiling !== void 0 && !(Number.isFinite(config.costPerTaskCeiling) && config.costPerTaskCeiling > 0)) throw new Error("HeldOutGate: costPerTaskCeiling must be a positive finite number");
|
|
10250
10309
|
this.costPerTaskCeiling = config.costPerTaskCeiling;
|
|
10251
10310
|
}
|
|
@@ -10295,6 +10354,21 @@ var HeldOutGate = class {
|
|
|
10295
10354
|
searchCoverage,
|
|
10296
10355
|
holdoutCoverage
|
|
10297
10356
|
};
|
|
10357
|
+
/** Evidence for the early exits, where no CI was computed at all. The
|
|
10358
|
+
* statistic that WOULD have decided is still reported so a caller can see
|
|
10359
|
+
* which branch their data lands in without producing a verdict. */
|
|
10360
|
+
const earlyShape = this.chooseShape(beforeHoldout, afterHoldout);
|
|
10361
|
+
const noCiEvidence = {
|
|
10362
|
+
...commonEvidence,
|
|
10363
|
+
medianPairedDelta: productiveRuns > 0 ? medianDelta(beforeHoldout, afterHoldout) : null,
|
|
10364
|
+
deltaStatistic: earlyShape.statistic,
|
|
10365
|
+
decidingDelta: null,
|
|
10366
|
+
pairedCI: null,
|
|
10367
|
+
pairedPValue: null,
|
|
10368
|
+
mcnemar: null,
|
|
10369
|
+
binaryScale: earlyShape.binaryScale,
|
|
10370
|
+
tieFraction: earlyShape.tieFraction
|
|
10371
|
+
};
|
|
10298
10372
|
const missingSplitScores = [
|
|
10299
10373
|
candidateSearch.length === 0 ? "candidate search" : null,
|
|
10300
10374
|
candidateHoldout.length === 0 ? "candidate holdout" : null,
|
|
@@ -10305,12 +10379,7 @@ var HeldOutGate = class {
|
|
|
10305
10379
|
promote: false,
|
|
10306
10380
|
candidateId,
|
|
10307
10381
|
baselineId,
|
|
10308
|
-
evidence:
|
|
10309
|
-
...commonEvidence,
|
|
10310
|
-
medianPairedDelta: productiveRuns > 0 ? medianDelta(beforeHoldout, afterHoldout) : null,
|
|
10311
|
-
pairedCI: null,
|
|
10312
|
-
pairedPValue: null
|
|
10313
|
-
},
|
|
10382
|
+
evidence: noCiEvidence,
|
|
10314
10383
|
reason: `missing_split_scores: ${missingSplitScores.join(", ")} score evidence is absent`,
|
|
10315
10384
|
rejectionCode: "missing_split_scores"
|
|
10316
10385
|
};
|
|
@@ -10319,12 +10388,7 @@ var HeldOutGate = class {
|
|
|
10319
10388
|
promote: false,
|
|
10320
10389
|
candidateId,
|
|
10321
10390
|
baselineId,
|
|
10322
|
-
evidence:
|
|
10323
|
-
...commonEvidence,
|
|
10324
|
-
medianPairedDelta: productiveRuns > 0 ? medianDelta(beforeHoldout, afterHoldout) : null,
|
|
10325
|
-
pairedCI: null,
|
|
10326
|
-
pairedPValue: null
|
|
10327
|
-
},
|
|
10391
|
+
evidence: noCiEvidence,
|
|
10328
10392
|
reason: `incomplete_coverage: ${shortfalls.map(([split, cov]) => describeCoverage(split, cov)).join("; ")} — required ${fmt(this.minCoverage)} of every dealt item scored on both arms. A missing score is missing data, not an absent item: the gate will not decide on the subset that answered`,
|
|
10329
10393
|
rejectionCode: "incomplete_coverage"
|
|
10330
10394
|
};
|
|
@@ -10332,51 +10396,56 @@ var HeldOutGate = class {
|
|
|
10332
10396
|
promote: false,
|
|
10333
10397
|
candidateId,
|
|
10334
10398
|
baselineId,
|
|
10335
|
-
evidence:
|
|
10336
|
-
...commonEvidence,
|
|
10337
|
-
medianPairedDelta: productiveRuns > 0 ? medianDelta(beforeHoldout, afterHoldout) : null,
|
|
10338
|
-
pairedCI: null,
|
|
10339
|
-
pairedPValue: null
|
|
10340
|
-
},
|
|
10399
|
+
evidence: noCiEvidence,
|
|
10341
10400
|
reason: `few_runs: ${productiveRuns} paired holdout observation(s) < min ${this.minProductiveRuns}`,
|
|
10342
10401
|
rejectionCode: "few_runs"
|
|
10343
10402
|
};
|
|
10344
10403
|
if (overfitGap === null || baselineOverfitGap === null) throw new Error("HeldOutGate: complete split scores did not produce overfit gaps");
|
|
10345
|
-
const
|
|
10404
|
+
const decision = decidePairedPromotion(beforeHoldout, afterHoldout, {
|
|
10405
|
+
threshold: this.pairedDeltaThreshold,
|
|
10346
10406
|
confidence: this.confidence,
|
|
10347
10407
|
resamples: this.resamples,
|
|
10348
|
-
statistic: "median",
|
|
10349
10408
|
seed: this.seed,
|
|
10350
|
-
|
|
10351
|
-
|
|
10409
|
+
minPairs: this.minProductiveRuns,
|
|
10410
|
+
statistic: this.deltaStatistic === "median" ? "median" : "mean"
|
|
10352
10411
|
});
|
|
10353
|
-
const
|
|
10412
|
+
const literalMedian = medianDelta(beforeHoldout, afterHoldout);
|
|
10354
10413
|
const wilcoxon = wilcoxonSignedRank(beforeHoldout, afterHoldout);
|
|
10414
|
+
const { delta: decidingDelta, low, high, label: deltaLabel } = decision;
|
|
10355
10415
|
const evidence = {
|
|
10356
10416
|
...commonEvidence,
|
|
10357
|
-
medianPairedDelta:
|
|
10417
|
+
medianPairedDelta: literalMedian,
|
|
10418
|
+
deltaStatistic: decision.statistic,
|
|
10419
|
+
decidingDelta,
|
|
10358
10420
|
pairedCI: {
|
|
10359
|
-
low
|
|
10360
|
-
high
|
|
10421
|
+
low,
|
|
10422
|
+
high
|
|
10361
10423
|
},
|
|
10362
|
-
pairedPValue: wilcoxon.p
|
|
10424
|
+
pairedPValue: wilcoxon.p,
|
|
10425
|
+
mcnemar: decision.mcnemar,
|
|
10426
|
+
binaryScale: decision.binaryScale,
|
|
10427
|
+
tieFraction: decision.tieFraction
|
|
10363
10428
|
};
|
|
10364
|
-
if (
|
|
10429
|
+
if (decision.indeterminate) return {
|
|
10365
10430
|
promote: false,
|
|
10366
10431
|
candidateId,
|
|
10367
10432
|
baselineId,
|
|
10368
10433
|
evidence,
|
|
10369
|
-
reason: `indeterminate_delta: paired holdout
|
|
10434
|
+
reason: `indeterminate_delta: ${decision.indeterminateCause}, so the paired holdout CI is [${fmt(low)}, ${fmt(high)}] and carries no direction — it cannot clear threshold ${fmt(this.pairedDeltaThreshold)} on evidence`,
|
|
10370
10435
|
rejectionCode: "indeterminate_delta"
|
|
10371
10436
|
};
|
|
10372
|
-
|
|
10373
|
-
|
|
10374
|
-
|
|
10375
|
-
|
|
10376
|
-
|
|
10377
|
-
|
|
10378
|
-
|
|
10379
|
-
|
|
10437
|
+
const exactTestVetoes = decision.exactTestVetoes;
|
|
10438
|
+
if (!decision.clearsThreshold || exactTestVetoes) {
|
|
10439
|
+
const detail = exactTestVetoes ? ` McNemar exact p=${decision.mcnemar?.pValue.toExponential(2)} does not reject at α=${fmt(1 - this.confidence)}.` : decision.methodDetail;
|
|
10440
|
+
return {
|
|
10441
|
+
promote: false,
|
|
10442
|
+
candidateId,
|
|
10443
|
+
baselineId,
|
|
10444
|
+
evidence,
|
|
10445
|
+
reason: `negative_delta: paired holdout ${deltaLabel} Δ=${fmt(decidingDelta)} CI=[${fmt(low)}, ${fmt(high)}] does not clear threshold ${fmt(this.pairedDeltaThreshold)}.${detail}`,
|
|
10446
|
+
rejectionCode: "negative_delta"
|
|
10447
|
+
};
|
|
10448
|
+
}
|
|
10380
10449
|
if (overfitGap > baselineOverfitGap + this.overfitGapThreshold) return {
|
|
10381
10450
|
promote: false,
|
|
10382
10451
|
candidateId,
|
|
@@ -10406,10 +10475,19 @@ var HeldOutGate = class {
|
|
|
10406
10475
|
candidateId,
|
|
10407
10476
|
baselineId,
|
|
10408
10477
|
evidence,
|
|
10409
|
-
reason: `promote: paired holdout
|
|
10478
|
+
reason: `promote: paired holdout ${deltaLabel} Δ=${fmt(decidingDelta)} CI=[${fmt(low)}, ${fmt(high)}] over ${productiveRuns} pairs; overfit gap candidate=${fmt(overfitGap)} vs baseline=${fmt(baselineOverfitGap)}; median cost candidate=$${fmt(medianCandidateCost)} vs baseline=$${fmt(medianBaselineCost)}`,
|
|
10410
10479
|
rejectionCode: null
|
|
10411
10480
|
};
|
|
10412
10481
|
}
|
|
10482
|
+
/**
|
|
10483
|
+
* Which paired statistic decides this data, and the shape facts behind it —
|
|
10484
|
+
* for the early exits, where no interval is computed. Delegates to
|
|
10485
|
+
* {@link pairedDecisionShape} so the reported shape can never disagree with
|
|
10486
|
+
* the shape the decision actually routes on.
|
|
10487
|
+
*/
|
|
10488
|
+
chooseShape(before, after) {
|
|
10489
|
+
return pairedDecisionShape(before, after, this.deltaStatistic === "median" ? "median" : "mean");
|
|
10490
|
+
}
|
|
10413
10491
|
};
|
|
10414
10492
|
function inferCandidateId(candidate, baselineKey) {
|
|
10415
10493
|
for (const run of candidate) if (run.candidateId && run.candidateId !== baselineKey) return run.candidateId;
|
|
@@ -12294,6 +12372,6 @@ function assertProductBenchmarkRun(runDir) {
|
|
|
12294
12372
|
return report;
|
|
12295
12373
|
}
|
|
12296
12374
|
//#endregion
|
|
12297
|
-
export { AGENT_PROFILE_KINDS, ATIF_SCHEMA_VERSION, ATTESTATION_ALGORITHM, AgentDriver, AgentEvalError, AgentProfileCellValidationError, AnalystRegistry, AxGepaSteeringOptimizer, BENCHMARK_SPLIT_SEED, BOOTSTRAP_GATE_MIN_N, BackendIntegrityError, BenchmarkRunner, BudgetBreachError, BudgetGuard, CODING_HARNESSES, CallbackResearcher, CaptureIntegrityError, ConfigError, ConvergenceTracker, CostAccountingIncompleteError, CostCallConflictError, CostCeilingReachedError, CostLedger, CostLedgerPersistenceError, CostReceiptCaptureError, CostReservationExceededError, CostTracker, CrossFamilyError, DEFAULT_AGENT_SLOS, DEFAULT_COMPLEXITY_WEIGHTS, DEFAULT_RULES as DEFAULT_FAILURE_RULES, DEFAULT_FINDERS, DEFAULT_HARNESS_OBJECTIVES, DEFAULT_MUTATION_PRIMITIVES, DEFAULT_MUTATORS, DEFAULT_PERMUTATIONS, DEFAULT_PR_REVIEW_SCORE_WEIGHTS, DEFAULT_REDACTION_RULES, DEFAULT_RED_TEAM_CORPUS, DEFAULT_RUN_SCORE_WEIGHTS, DEFAULT_SEVERITY_WEIGHTS, DEFAULT_TRACE_ANALYST_BUDGETS, DEFAULT_TRACE_ANALYST_KINDS, Dataset, DescriptionLengthGate, DockerSandboxDriver, DualAgentBench, ERROR_COUNT_PATTERNS, EvalTraceStore, ExperimentTracker, FAILURE_CLASSES, FAILURE_MODE_KIND_SPEC, FileSystemFeedbackTrajectoryStore, FileSystemRawProviderSink, FileSystemTraceStore, FindingsStore, HARBOR_IMPORT_GAP, HARNESS_NATIVE_MODEL, HeldOutGate, HoldoutAuditor, HoldoutLockedError, IMPROVEMENT_KIND_SPEC, INPUT_VALUE, INTENT_MATCH_JUDGE_VERSION, InMemoryFeedbackTrajectoryStore, InMemoryRawProviderSink, InMemoryTraceStore, InMemoryWorkspaceInspector, JudgeError, JudgeParseError, JudgeRunner, KNOWLEDGE_GAP_KIND_SPEC, KNOWLEDGE_POISONING_KIND_SPEC, LLM_CACHED_TOKENS, LLM_CACHED_TOKEN_ATTR_KEYS, LLM_CACHE_WRITE_TOKENS, LLM_CACHE_WRITE_TOKEN_ATTR_KEYS, LLM_CONTEXT_TOKENS, LLM_COST_ATTR_KEYS, LLM_COST_USD, LLM_INPUT_TOKENS, LLM_INPUT_TOKEN_ATTR_KEYS, LLM_MODEL_ATTR_KEYS, LLM_MODEL_NAME, LLM_OUTPUT_TOKENS, LLM_OUTPUT_TOKEN_ATTR_KEYS, LLM_REASONING_TOKENS, LLM_REASONING_TOKEN_ATTR_KEYS, LlmCallError, LlmClient, LlmResponseError, LlmRouteAssertionError, LockedJsonlAppender, MANN_WHITNEY_EXACT_MAX_STATES, MANN_WHITNEY_EXACT_MAX_WORK, MODEL_PRICING, MetricsCollector, ModelsUnreachableError, MultiLayerVerifier, Mutex, NoopRawProviderSink, NoopResearcher, NotFoundError, OPENINFERENCE_SPAN_KIND, OTEL_AGENT_EVAL_SCOPE, OUTPUT_VALUE, OtlpFileTraceStore, PairwiseSteeringOptimizer, ProductClient, PromptRegistry, REDACTION_VERSION, REFERENCE_EQUIVALENCE_INPUT_LIMITS, REFERENCE_EQUIVALENCE_JUDGE_VERSION, RESEARCH_REPORT_HARD_PAIR_FLOOR, ROLLOUT_SCHEMA, RUN_COST_ATTR_KEYS, ReplayCache, ReplayCacheMissError, ReplayError, RunCritic, RunIntegrityError, RunRecordValidationError, SEMANTIC_CONCEPT_JUDGE_VERSION, SKILL_USAGE_ANALYST, SPAN_KIND_ATTR_KEYS, SUPERVISOR_RUN_SCHEMA, SandboxHarness, ScenarioRegistry, SeatUnsetError, SingleBackendError, SkillUsageAnalyst, SpanNotFoundError, SubprocessSandboxDriver, TOOL_ARGS_CAPTURED, TOOL_LATENCY_MS, TOOL_NAME, TOOL_NAME_ATTR_KEYS, TRACE_ANALYST_ACTOR_DESCRIPTION, TRACE_ANALYST_ACTOR_DESCRIPTION_VERSION, TRACE_ANALYST_TRUNCATION_MARKER_PREFIX, TRACE_SCHEMA_VERSION, TokenCounter, TraceContractBuilder, TraceEmitter, TraceFileMissingError, TraceNotFoundError, UI_FINDING_SEVERITIES, UI_LENSES, UNIVERSAL_FINDERS, ValidationError, VerificationError, WILCOXON_EXACT_MAX_N, acquisitionPlansForKnowledgeGaps, agentProfileCellHashMaterial, agentProfileCellKey, agentProfileHash, agentProfileId, agentProfileModelId, agentVisibleFields, aggregateJudgeVerdicts, aggregateLlm, aggregatePrReviewScore, aggregateRunScore, allCriticalPassed, analyzeAntiSlop, analyzeSeries, analyzeSupervisorRun, analyzeSupervisorRunSources, analyzeTraces, appendScorecard, applyLlmSpanOtlpAttributes, applyToolSpanOtlpAttributes, argHash, asNumber, asString, assertCapabilityHeadroom, assertCrossFamily, assertLlmRoute, assertMinted, assertMintedLines, assertModelsServed, assertNoHiddenLeak, assertProductBenchmarkRun, assertRealAgentReceipts, assertRealBackend, assertReleaseConfidence, assertRolloutLine, assertRunAgentProfileCell, assertRunCaptured, assertSingleBackend, assignFeedbackSplit, assignHeldOutTag, attachCostToReport, attest, attributeCounterfactuals, backoffMs, deterministicSplit as benchmarkDeterministicSplit, benchmarks_exports as benchmarks, benjaminiHochberg, bisect, blendHeldout, blockingKnowledgeEval, bonferroni, bootstrapCi, buildAgentInterfaceProfileCell, buildAgentProfileCell, buildDefaultAnalystRegistry, buildDriverSystemPrompt, buildProductBenchmarkManifest, buildReflectionPrompt, buildReviewerPrompt, buildTraceAnalystTools, buildTraceInsightContext, buildTraceInsightPrompt, buildTrajectory, buildWorkerDriverSystemPrompt, byteLengthRange, cachedJudge, calibrateJudge, calibrateJudgeContinuous, callLlm, callLlmJson, canaryLeakView, canonicalJson, canonicalize, capabilityHeadroom, captureFetchToRawSink, causalAttribution, checkBehavioralCanary, checkCanaries, checkSlos, checkTraceContracts, clamp01, classifyFailure, classifyOtlpSpanRole, classifyTreatment, claudeCodeSupervisorRunReader, cliffsDelta, clusteredPairedBinary, cohensD, collectionPreserved, commentsForSource, commitBisect, comparePairedArms, compareReferenceReplay, compareToBaseline, compilerJudge, completionVerdict, composeParsers, composeValidators, computeExperimentStats, computeFindingId, computeToolUseMetrics, computeTraceMetrics, confidenceInterval, containsAll, contentHash, contextInputTokens, continuousAgreement, contractJudge, controlFailureClassFromVerification, controlRunToFeedbackTrajectory, controlRunToRunRecord, convertTraceStoresToOtlp, corpusInterRaterAgreement, corpusInterRaterAgreementFromJudgeScores, costForTokenPricing, costForUsage, costReceiptFromLlm, costReceiptFromLlmError, costReport, createAnalystAi, createAntiSlopJudge, createChatClient, createDefaultReviewer, createFeedbackTrajectory, createIntentMatchJudge, createLlmCorrectnessChecker, createLlmReviewer, createOtelExporter, createOtelTracingStore, createReferenceEquivalenceJudge, createReplayFetch, createSandboxPool, createSemanticConceptJudge, createTokenRecallChecker, createTraceAnalystKind, crossTraceDiff, crowdingDistance, dataDescriptionBits, decideNextUserTurn, decideReferenceReplayPromotion, decideReferenceReplayRunPromotion, defaultBlendWeights, defaultIsMaterial, defaultProviderRedactor, defaultReferenceReplayMatcher, defaultTraceInsightPanel, deployGateLayer, describeTraceInsightScope, diffFindings, diffScorecard, discoverPersonas, distillPlaybook, domainEvidencePattern, dominates, eProcess, ensembleJudge, errorStreakDetector, estimateCost, estimateTokens, evaluateActionPolicy, evaluateContract, evaluateHypothesis, evaluateInterimReleaseConfidence, evaluateOracles, evaluateReleaseConfidence, evaluateTraceContract, executeScenario, expandProfileAxes, expectAgent, exportProductBenchmark, exportProductBenchmarkRuns, exportRewardModel, exportRunAsOtlp, extractAssetUrls, extractErrorCount, extractOtlpAttributes, extractProducedState, extractUsage, extractUsageFromResponse, extractUsageFromSse, feedbackTrajectoriesToDatasetScenarios, feedbackTrajectoriesToOptimizerRows, feedbackTrajectoryToDatasetScenario, feedbackTrajectoryToOptimizerRow, fileContains, fileExists, fileExperimentStore, fileVerdictCache, findAutoMatchNoExpectation, findConstructorCwdDropped, findFallbackToPass, findLiteralTruePass, findProductBenchmarkArtifacts, findSkipCountsAsPass, firstNumberAttr, firstStringAttr, flattenOtlpExportToNdjson, flowLayer, fnv1a32, formatBenchmarkReport, formatDriverReport, formatFindings, formatScorecardDiff, fromHarborTrajectory, gainHistogram, gateTreatmentApplied, gateTreatmentFromMetrics, gateTreatmentFromSpans, gateTreatmentFromToolSpans, ghCliClient, gitProvenanceReader, precision as goldenPrecision, gradeOnHidden, gradeSemanticStatus, groupBy, groupRunsByAgentProfileCell, harnessAxisOf, hasCapturedToolArgs, hashContent, hashJson, hashScenarios, hashToUnit, hiddenGrade, holm, htmlContainsElement, httpGithubClient, improvementVerdict, inMemoryExperimentStore, inMemoryReferenceReplayStore, inMemoryReviewStore, inMemoryRunRecordBackend, inMemoryVerdictCache, inferDomainKeywords, inferOtlpKind, interRaterReliability, interpretCliffs, iqr, isHiddenDestination, isJudgeSpan, isLlmSpan, isModelPriced, isOtelConfigured, isOtlpModelCall, isRealnessGated, isRetrievalSpan, isRolloutLine, isRunRecord, isSandboxSpan, isToolSpan, isTrainableSplit, isTransientLlmError, isUnavailable, iterateRawCalls, jestTestParser, jsonHasKeys, jsonShape, jsonlReferenceReplayStore, jsonlReviewStore, jsonlRunRecordBackend, judgeFamily, judgeReplayGate, judgeSpans, keyPreserved, knowledgeReadinessTracePayload, leaderboard, linterJudge, llmJudge, llmSpanFromProvider, llmSpans, loadScorecard, loadScorerFromGrader, localCommandRunner, lowercaseMutator, makeEvalTools, makeFinding, makeProposalFinding, mannWhitneyU, mapConcurrent, matchGoldens, matchSpan, maximumChargeForLlmRequest, mcnemar, mcnemarPower, mcnemarRequiredN, mergeLayerResults, mergeSteeringBundle, minimumPairsForPairedDeltaTest, mintRolloutRows, modelDescriptionBits, modelHasSnapshot, modelPriceKey, mulberry32, multiToolchainLayer, noProgressDetector, normalCdf, normalizeScores, notBlocked, objectiveEval, observeAll, observedScore, observedSplitScore, otelRunCompleteHook, otlpRowsToRunRecords, otlpRowsToTraceRunRecords, otlpToRunRecords, otlpToTraceRunRecords, pairArms, pairRunRecords, pairedBootstrap, pairedCohensDz, pairedDeltaTest, pairedEvalueSequence, pairedMde, pairedRiskDifference, pairedSignTest, pairedTTest, paraphraseRobustness, paraphraseRobustnessScenarios, paretoChart, paretoFrontier, paretoFrontierWithCrowding, parseCorrectnessResponse, parseFeedbackTrajectoriesJsonl, parseReflectionResponse, parseRunRecordSafe, parseRuntimeTrajectoryHookEvent, partialCredit, partitionHeldOut, passAtK, passOrthogonality, pearsonR, pixelDeltaRatio, planTraceInsightQuestions, politenessPrefixMutator, positionalBias, preflightModels, printDriverSummary, probeLlm, productBenchmarkIntegrityFailures, productBenchmarkMutableSurfaces, productBenchmarkRepoIdentity, productBenchmarkSplits, profile_exports as profile, projectOtlpFlatLine, projectRuntimeTrajectoryEvidence, promptBisect, proposeSynthesisTargets, providerFromBaseUrl, pytestTestParser, ranks, readClaudeCodeSupervisorRun, readOtlpStatus, readProductBenchmarkManifest, readProductBenchmarkRecords, recordRuns, recordRunsToScorecard, redTeamDataset, redTeamReport, redactString, redactValue, referenceReplayRunsToSteeringRows, referenceReplayScenarioToRunScore, regexMatch, regexMatches, relabelImportedSplit, renderMarkdownReport, renderPlaybookMarkdown, renderPreferenceMemoryMarkdown, renderPriorFindings, renderReleaseReport, renderSteeringText, renderSupervisorRunHeadline, renderSupervisorRunMarkdown, renderUpstreamFindings, repeatedActionDetector, replayFeedbackTrajectories, replayFeedbackTrajectory, replayScorerOverCorpus, replayTraceThroughJudge, requireAgentProfileCell, requiredPairedSampleSize, requiredSampleSize, researchReport, resolveModelPricing, resolveSeat, rollupSupervisorRuns, roundTripRunRecord, routeFields, rowCount, rowWhere, runAgentControlLoop, runAssertions, runBehavioralCanaries, runCanaries, runCounterfactual, runE2EWorkflow, runEvalCampaign, runExpectations, runFailureClass, runHarnessExperiment, runIntentMatchJudge, runJudgeFleet, runKeywordCoverageJudge, runKeywordCoverageJudgeUrl, runLiveProof, runProposeReview, runProposeReviewAsControlLoop, runRecordToProductBenchmarkRecord, runReferenceEquivalenceJudge, runReferenceReplay, runScore, runSelfPlay, runSemanticConceptJudge, runTaskScore, runTestGradedScenario, runsForScenario, scalarScore, scanForMuffledGates, scoreContinuity, scoreFromEvals, scoreKnowledgeReadiness, scoreOrigin, scorePrReviewComments, scorePrReviewSource, scoreRedTeamOutput, scoreReferenceReplay, scoreTraceInsightReadiness, seatPresets, securityJudge, selectHarnessVariant, selfPreference, sentenceReorderMutator, serializeFeedbackTrajectoriesJsonl, showMeasured, signManifest, spearmanR, statusAdvanced, stopOnNoProgress, stopOnRepeatedAction, stringField, stripFencedJson, studentTCdf, subjectiveEval, summarizeAgentReceiptIntegrity, summarizeBackendIntegrity, summarizeHarnessResults, summarizePrReviewBenchmark, summarizePreferenceMemory, summaryTable, supervisorRunRolloutLines, testJudge, textInSnapshot, throwIfRunIncomplete, toAgentProfileJson, toHarborTrajectories, toHarborTrajectory, toJsonl, toLangfuseEnvelope, toOpenAiTool, toPrometheusText, toRewardRows, toSftRows, tokenizeDomainWords, toolNamesForRun, toolSpans, traceAnalystFunctionGroup, traceAnalystOnRunComplete, traceContract, traceJudge, traceJudgeEnsemble, traceSpanKindToOpenInferenceKind, tracedAnalyzeTraces, trainingReward, trainingScore, typoMutator, unmintableReasons, urlContains, userQuestionsForKnowledgeGaps, validateAgentProfileCell, validateProductBenchmarkManifest, validateProductBenchmarkRecord, validateProductBenchmarkRun, validateRolloutLine, validateRunRecord, verbosityBias, verifyAgentProfileCell, verifyAttestation, verifyCompletion, verifyManifest, visualDiff, viteDeployRunner, vitestTestParser, weightedComposite, weightedMean, weightedRecall, welchsTTest, whitespaceCollapseMutator, wilcoxonSignedRank, wilson, withAssignedFeedbackSplit, withHeldoutBlend, withJudgeRetry, withOtelPipeline, wranglerDeployRunner, writeSupervisorRunReport };
|
|
12375
|
+
export { AGENT_PROFILE_KINDS, ATIF_SCHEMA_VERSION, ATTESTATION_ALGORITHM, AgentDriver, AgentEvalError, AgentProfileCellValidationError, AnalystRegistry, AxGepaSteeringOptimizer, BENCHMARK_SPLIT_SEED, BOOTSTRAP_GATE_MIN_N, BackendIntegrityError, BenchmarkRunner, BudgetBreachError, BudgetGuard, CODING_HARNESSES, CallbackResearcher, CaptureIntegrityError, ConfigError, ConvergenceTracker, CostAccountingIncompleteError, CostCallConflictError, CostCeilingReachedError, CostLedger, CostLedgerPersistenceError, CostReceiptCaptureError, CostReservationExceededError, CostTracker, CrossFamilyError, DECISION_PAIRED_DELTA_STATISTIC, DEFAULT_AGENT_SLOS, DEFAULT_COMPLEXITY_WEIGHTS, DEFAULT_RULES as DEFAULT_FAILURE_RULES, DEFAULT_FINDERS, DEFAULT_HARNESS_OBJECTIVES, DEFAULT_MUTATION_PRIMITIVES, DEFAULT_MUTATORS, DEFAULT_PERMUTATIONS, DEFAULT_PR_REVIEW_SCORE_WEIGHTS, DEFAULT_REDACTION_RULES, DEFAULT_RED_TEAM_CORPUS, DEFAULT_RUN_SCORE_WEIGHTS, DEFAULT_SEVERITY_WEIGHTS, DEFAULT_TRACE_ANALYST_BUDGETS, DEFAULT_TRACE_ANALYST_KINDS, Dataset, DescriptionLengthGate, DockerSandboxDriver, DualAgentBench, ERROR_COUNT_PATTERNS, EvalTraceStore, ExperimentTracker, FAILURE_CLASSES, FAILURE_MODE_KIND_SPEC, FileSystemFeedbackTrajectoryStore, FileSystemRawProviderSink, FileSystemTraceStore, FindingsStore, HARBOR_IMPORT_GAP, HARNESS_NATIVE_MODEL, HeldOutGate, HoldoutAuditor, HoldoutLockedError, IMPROVEMENT_KIND_SPEC, INPUT_VALUE, INTENT_MATCH_JUDGE_VERSION, InMemoryFeedbackTrajectoryStore, InMemoryRawProviderSink, InMemoryTraceStore, InMemoryWorkspaceInspector, JudgeError, JudgeParseError, JudgeRunner, KNOWLEDGE_GAP_KIND_SPEC, KNOWLEDGE_POISONING_KIND_SPEC, LLM_CACHED_TOKENS, LLM_CACHED_TOKEN_ATTR_KEYS, LLM_CACHE_WRITE_TOKENS, LLM_CACHE_WRITE_TOKEN_ATTR_KEYS, LLM_CONTEXT_TOKENS, LLM_COST_ATTR_KEYS, LLM_COST_USD, LLM_INPUT_TOKENS, LLM_INPUT_TOKEN_ATTR_KEYS, LLM_MODEL_ATTR_KEYS, LLM_MODEL_NAME, LLM_OUTPUT_TOKENS, LLM_OUTPUT_TOKEN_ATTR_KEYS, LLM_REASONING_TOKENS, LLM_REASONING_TOKEN_ATTR_KEYS, LlmCallError, LlmClient, LlmResponseError, LlmRouteAssertionError, LockedJsonlAppender, MANN_WHITNEY_EXACT_MAX_STATES, MANN_WHITNEY_EXACT_MAX_WORK, MODEL_PRICING, MetricsCollector, ModelsUnreachableError, MultiLayerVerifier, Mutex, NoopRawProviderSink, NoopResearcher, NotFoundError, OPENINFERENCE_SPAN_KIND, OTEL_AGENT_EVAL_SCOPE, OUTPUT_VALUE, OtlpFileTraceStore, PairwiseSteeringOptimizer, ProductClient, PromptRegistry, REDACTION_VERSION, REFERENCE_EQUIVALENCE_INPUT_LIMITS, REFERENCE_EQUIVALENCE_JUDGE_VERSION, RESEARCH_REPORT_HARD_PAIR_FLOOR, ROLLOUT_SCHEMA, RUN_COST_ATTR_KEYS, ReplayCache, ReplayCacheMissError, ReplayError, RunCritic, RunIntegrityError, RunRecordValidationError, SEMANTIC_CONCEPT_JUDGE_VERSION, SKILL_USAGE_ANALYST, SPAN_KIND_ATTR_KEYS, SUPERVISOR_RUN_SCHEMA, SandboxHarness, ScenarioRegistry, SeatUnsetError, SingleBackendError, SkillUsageAnalyst, SpanNotFoundError, SubprocessSandboxDriver, TOOL_ARGS_CAPTURED, TOOL_LATENCY_MS, TOOL_NAME, TOOL_NAME_ATTR_KEYS, TRACE_ANALYST_ACTOR_DESCRIPTION, TRACE_ANALYST_ACTOR_DESCRIPTION_VERSION, TRACE_ANALYST_TRUNCATION_MARKER_PREFIX, TRACE_SCHEMA_VERSION, TokenCounter, TraceContractBuilder, TraceEmitter, TraceFileMissingError, TraceNotFoundError, UI_FINDING_SEVERITIES, UI_LENSES, UNIVERSAL_FINDERS, ValidationError, VerificationError, WILCOXON_EXACT_MAX_N, acquisitionPlansForKnowledgeGaps, agentProfileCellHashMaterial, agentProfileCellKey, agentProfileHash, agentProfileId, agentProfileModelId, agentVisibleFields, aggregateJudgeVerdicts, aggregateLlm, aggregatePrReviewScore, aggregateRunScore, allCriticalPassed, analyzeAntiSlop, analyzeSeries, analyzeSupervisorRun, analyzeSupervisorRunSources, analyzeTraces, appendScorecard, applyLlmSpanOtlpAttributes, applyToolSpanOtlpAttributes, argHash, asNumber, asString, assertCapabilityHeadroom, assertCrossFamily, assertLlmRoute, assertMinted, assertMintedLines, assertModelsServed, assertNoHiddenLeak, assertProductBenchmarkRun, assertRealAgentReceipts, assertRealBackend, assertReleaseConfidence, assertRolloutLine, assertRunAgentProfileCell, assertRunCaptured, assertSingleBackend, assignFeedbackSplit, assignHeldOutTag, attachCostToReport, attest, attributeCounterfactuals, backoffMs, deterministicSplit as benchmarkDeterministicSplit, benchmarks_exports as benchmarks, benjaminiHochberg, bisect, blendHeldout, blockingKnowledgeEval, bonferroni, bootstrapCi, buildAgentInterfaceProfileCell, buildAgentProfileCell, buildDefaultAnalystRegistry, buildDriverSystemPrompt, buildProductBenchmarkManifest, buildReflectionPrompt, buildReviewerPrompt, buildTraceAnalystTools, buildTraceInsightContext, buildTraceInsightPrompt, buildTrajectory, buildWorkerDriverSystemPrompt, byteLengthRange, cachedJudge, calibrateJudge, calibrateJudgeContinuous, callLlm, callLlmJson, canaryLeakView, canonicalJson, canonicalize, capabilityHeadroom, captureFetchToRawSink, causalAttribution, checkBehavioralCanary, checkCanaries, checkSlos, checkTraceContracts, clamp01, classifyFailure, classifyOtlpSpanRole, classifyTreatment, claudeCodeSupervisorRunReader, cliffsDelta, clusteredPairedBinary, cohensD, collectionPreserved, commentsForSource, commitBisect, comparePairedArms, compareReferenceReplay, compareToBaseline, compilerJudge, completionVerdict, composeParsers, composeValidators, computeExperimentStats, computeFindingId, computeToolUseMetrics, computeTraceMetrics, confidenceInterval, containsAll, contentHash, contextInputTokens, continuousAgreement, contractJudge, controlFailureClassFromVerification, controlRunToFeedbackTrajectory, controlRunToRunRecord, convertTraceStoresToOtlp, corpusInterRaterAgreement, corpusInterRaterAgreementFromJudgeScores, costForTokenPricing, costForUsage, costReceiptFromLlm, costReceiptFromLlmError, costReport, createAnalystAi, createAntiSlopJudge, createChatClient, createDefaultReviewer, createFeedbackTrajectory, createIntentMatchJudge, createLlmCorrectnessChecker, createLlmReviewer, createOtelExporter, createOtelTracingStore, createReferenceEquivalenceJudge, createReplayFetch, createSandboxPool, createSemanticConceptJudge, createTokenRecallChecker, createTraceAnalystKind, crossTraceDiff, crowdingDistance, dataDescriptionBits, decideNextUserTurn, decidePairedPromotion, decideReferenceReplayPromotion, decideReferenceReplayRunPromotion, defaultBlendWeights, defaultIsMaterial, defaultProviderRedactor, defaultReferenceReplayMatcher, defaultTraceInsightPanel, deployGateLayer, describeTraceInsightScope, diffFindings, diffScorecard, discoverPersonas, distillPlaybook, domainEvidencePattern, dominates, eProcess, ensembleJudge, errorStreakDetector, estimateCost, estimateTokens, evaluateActionPolicy, evaluateContract, evaluateHypothesis, evaluateInterimReleaseConfidence, evaluateOracles, evaluateReleaseConfidence, evaluateTraceContract, executeScenario, expandProfileAxes, expectAgent, exportProductBenchmark, exportProductBenchmarkRuns, exportRewardModel, exportRunAsOtlp, extractAssetUrls, extractErrorCount, extractOtlpAttributes, extractProducedState, extractUsage, extractUsageFromResponse, extractUsageFromSse, feedbackTrajectoriesToDatasetScenarios, feedbackTrajectoriesToOptimizerRows, feedbackTrajectoryToDatasetScenario, feedbackTrajectoryToOptimizerRow, fileContains, fileExists, fileExperimentStore, fileVerdictCache, findAutoMatchNoExpectation, findConstructorCwdDropped, findFallbackToPass, findLiteralTruePass, findProductBenchmarkArtifacts, findSkipCountsAsPass, firstNumberAttr, firstStringAttr, flattenOtlpExportToNdjson, flowLayer, fnv1a32, formatBenchmarkReport, formatDriverReport, formatFindings, formatScorecardDiff, fromHarborTrajectory, gainHistogram, gateTreatmentApplied, gateTreatmentFromMetrics, gateTreatmentFromSpans, gateTreatmentFromToolSpans, ghCliClient, gitProvenanceReader, precision as goldenPrecision, gradeOnHidden, gradeSemanticStatus, groupBy, groupRunsByAgentProfileCell, harnessAxisOf, hasCapturedToolArgs, hashContent, hashJson, hashScenarios, hashToUnit, hiddenGrade, holm, htmlContainsElement, httpGithubClient, improvementVerdict, inMemoryExperimentStore, inMemoryReferenceReplayStore, inMemoryReviewStore, inMemoryRunRecordBackend, inMemoryVerdictCache, inferDomainKeywords, inferOtlpKind, interRaterReliability, interpretCliffs, iqr, isBinaryOutcomeVector, isHiddenDestination, isJudgeSpan, isLlmSpan, isModelPriced, isOtelConfigured, isOtlpModelCall, isRealnessGated, isRetrievalSpan, isRolloutLine, isRunRecord, isSandboxSpan, isToolSpan, isTrainableSplit, isTransientLlmError, isUnavailable, iterateRawCalls, jestTestParser, jsonHasKeys, jsonShape, jsonlReferenceReplayStore, jsonlReviewStore, jsonlRunRecordBackend, judgeFamily, judgeReplayGate, judgeSpans, keyPreserved, knowledgeReadinessTracePayload, leaderboard, linterJudge, llmJudge, llmSpanFromProvider, llmSpans, loadScorecard, loadScorerFromGrader, localCommandRunner, lowercaseMutator, makeEvalTools, makeFinding, makeProposalFinding, mannWhitneyU, mapConcurrent, matchGoldens, matchSpan, maximumChargeForLlmRequest, mcnemar, mcnemarPower, mcnemarRequiredN, mergeLayerResults, mergeSteeringBundle, minimumPairsForPairedDeltaTest, mintRolloutRows, modelDescriptionBits, modelHasSnapshot, modelPriceKey, mulberry32, multiToolchainLayer, noProgressDetector, normalCdf, normalizeScores, notBlocked, objectiveEval, observeAll, observedScore, observedSplitScore, otelRunCompleteHook, otlpRowsToRunRecords, otlpRowsToTraceRunRecords, otlpToRunRecords, otlpToTraceRunRecords, pairArms, pairRunRecords, pairedBinaryScale, pairedBootstrap, pairedCohensDz, pairedDecisionShape, pairedDeltaTest, pairedDeltaTieFraction, pairedEvalueSequence, pairedMde, pairedRiskDifference, pairedRiskDifferenceExact, pairedRiskDifferenceScore, pairedSignTest, pairedTTest, paraphraseRobustness, paraphraseRobustnessScenarios, paretoChart, paretoFrontier, paretoFrontierWithCrowding, parseCorrectnessResponse, parseFeedbackTrajectoriesJsonl, parseReflectionResponse, parseRunRecordSafe, parseRuntimeTrajectoryHookEvent, partialCredit, partitionHeldOut, passAtK, passOrthogonality, pearsonR, pixelDeltaRatio, planTraceInsightQuestions, politenessPrefixMutator, positionalBias, preflightModels, printDriverSummary, probeLlm, productBenchmarkIntegrityFailures, productBenchmarkMutableSurfaces, productBenchmarkRepoIdentity, productBenchmarkSplits, profile_exports as profile, projectOtlpFlatLine, projectRuntimeTrajectoryEvidence, promptBisect, proposeSynthesisTargets, providerFromBaseUrl, pytestTestParser, ranks, readClaudeCodeSupervisorRun, readOtlpStatus, readProductBenchmarkManifest, readProductBenchmarkRecords, recordRuns, recordRunsToScorecard, redTeamDataset, redTeamReport, redactString, redactValue, referenceReplayRunsToSteeringRows, referenceReplayScenarioToRunScore, regexMatch, regexMatches, relabelImportedSplit, renderMarkdownReport, renderPlaybookMarkdown, renderPreferenceMemoryMarkdown, renderPriorFindings, renderReleaseReport, renderSteeringText, renderSupervisorRunHeadline, renderSupervisorRunMarkdown, renderUpstreamFindings, repeatedActionDetector, replayFeedbackTrajectories, replayFeedbackTrajectory, replayScorerOverCorpus, replayTraceThroughJudge, requireAgentProfileCell, requiredPairedSampleSize, requiredSampleSize, researchReport, resolveModelPricing, resolveSeat, rollupSupervisorRuns, roundTripRunRecord, routeFields, rowCount, rowWhere, runAgentControlLoop, runAssertions, runBehavioralCanaries, runCanaries, runCounterfactual, runE2EWorkflow, runEvalCampaign, runExpectations, runFailureClass, runHarnessExperiment, runIntentMatchJudge, runJudgeFleet, runKeywordCoverageJudge, runKeywordCoverageJudgeUrl, runLiveProof, runProposeReview, runProposeReviewAsControlLoop, runRecordToProductBenchmarkRecord, runReferenceEquivalenceJudge, runReferenceReplay, runScore, runSelfPlay, runSemanticConceptJudge, runTaskScore, runTestGradedScenario, runsForScenario, scalarScore, scanForMuffledGates, scoreContinuity, scoreFromEvals, scoreKnowledgeReadiness, scoreOrigin, scorePrReviewComments, scorePrReviewSource, scoreRedTeamOutput, scoreReferenceReplay, scoreTraceInsightReadiness, seatPresets, securityJudge, selectHarnessVariant, selfPreference, sentenceReorderMutator, serializeFeedbackTrajectoriesJsonl, showMeasured, signManifest, spearmanR, statusAdvanced, stopOnNoProgress, stopOnRepeatedAction, stringField, stripFencedJson, studentTCdf, subjectiveEval, summarizeAgentReceiptIntegrity, summarizeBackendIntegrity, summarizeHarnessResults, summarizePrReviewBenchmark, summarizePreferenceMemory, summaryTable, supervisorRunRolloutLines, testJudge, textInSnapshot, throwIfRunIncomplete, toAgentProfileJson, toHarborTrajectories, toHarborTrajectory, toJsonl, toLangfuseEnvelope, toOpenAiTool, toPrometheusText, toRewardRows, toSftRows, tokenizeDomainWords, toolNamesForRun, toolSpans, traceAnalystFunctionGroup, traceAnalystOnRunComplete, traceContract, traceJudge, traceJudgeEnsemble, traceSpanKindToOpenInferenceKind, tracedAnalyzeTraces, trainingReward, trainingScore, typoMutator, unmintableReasons, urlContains, userQuestionsForKnowledgeGaps, validateAgentProfileCell, validateProductBenchmarkManifest, validateProductBenchmarkRecord, validateProductBenchmarkRun, validateRolloutLine, validateRunRecord, verbosityBias, verifyAgentProfileCell, verifyAttestation, verifyCompletion, verifyManifest, visualDiff, viteDeployRunner, vitestTestParser, weightedComposite, weightedMean, weightedRecall, welchsTTest, whitespaceCollapseMutator, wilcoxonSignedRank, wilson, withAssignedFeedbackSplit, withHeldoutBlend, withJudgeRetry, withOtelPipeline, wranglerDeployRunner, writeSupervisorRunReport };
|
|
12298
12376
|
|
|
12299
12377
|
//# sourceMappingURL=index.js.map
|