@tangle-network/agent-eval 0.144.6 → 0.144.7
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +24 -0
- package/README.md +2 -0
- package/dist/{benchmark-J9Qe6j2_.d.ts → agent-profile-CgDTo40f.d.ts} +91 -4
- package/dist/agent-profile-CgDTo40f.d.ts.map +1 -0
- package/dist/analyst/index.d.ts +470 -88
- package/dist/analyst/index.d.ts.map +1 -1
- package/dist/analyst/index.js +24 -5
- package/dist/analyst/index.js.map +1 -1
- package/dist/{analyze-runs-DWIvOAGk.js → analyze-runs-BNNK7irB.js} +3 -3
- package/dist/{analyze-runs-DWIvOAGk.js.map → analyze-runs-BNNK7irB.js.map} +1 -1
- package/dist/{baseline-D_fT6277.d.ts → baseline-CavEbRyH.d.ts} +2 -23
- package/dist/baseline-CavEbRyH.d.ts.map +1 -0
- package/dist/{benchmark-command-CQd78YHt.js → benchmark-command-BCafwNrf.js} +662 -605
- package/dist/benchmark-command-BCafwNrf.js.map +1 -0
- package/dist/benchmarks/index.d.ts +1 -1
- package/dist/benchmarks/index.js +1 -1
- package/dist/{benchmarks-BEOkuvIg.js → benchmarks-CDSolHq7.js} +4 -4
- package/dist/{benchmarks-BEOkuvIg.js.map → benchmarks-CDSolHq7.js.map} +1 -1
- package/dist/campaign/index.d.ts +7 -5
- package/dist/campaign/index.js +5 -3
- package/dist/{campaign-CXsdyym7.js → campaign-Tdy3h62h.js} +17 -301
- package/dist/campaign-Tdy3h62h.js.map +1 -0
- package/dist/cli.js +2 -2
- package/dist/{client-0JI64ovJ.d.ts → client-DjXROWpx.d.ts} +3 -3
- package/dist/{client-0JI64ovJ.d.ts.map → client-DjXROWpx.d.ts.map} +1 -1
- package/dist/{completion-verifier-CBiee74w.d.ts → completion-verifier-foUCLif_.d.ts} +5 -5
- package/dist/{completion-verifier-CBiee74w.d.ts.map → completion-verifier-foUCLif_.d.ts.map} +1 -1
- package/dist/contract/index.d.ts +9 -8
- package/dist/contract/index.d.ts.map +1 -1
- package/dist/contract/index.js +7 -6
- package/dist/contract/index.js.map +1 -1
- package/dist/control.d.ts +2 -2
- package/dist/control.js +1 -1
- package/dist/counterfactual-CWPTrMH7.js +126 -0
- package/dist/counterfactual-CWPTrMH7.js.map +1 -0
- package/dist/counterfactual-CxmxAONP.d.ts +72 -0
- package/dist/counterfactual-CxmxAONP.d.ts.map +1 -0
- package/dist/{default-registry-J9m-_tya.d.ts → default-registry-BZhStdjl.d.ts} +6 -5
- package/dist/default-registry-BZhStdjl.d.ts.map +1 -0
- package/dist/{default-registry-Dta70shL.js → default-registry-BaQXW1Ow.js} +2 -2
- package/dist/{default-registry-Dta70shL.js.map → default-registry-BaQXW1Ow.js.map} +1 -1
- package/dist/{dspy-rlm-engine-19FQEMBK.js → dspy-rlm-engine-BiN49gK6.js} +4 -3
- package/dist/{dspy-rlm-engine-19FQEMBK.js.map → dspy-rlm-engine-BiN49gK6.js.map} +1 -1
- package/dist/{tool-groups-CK0JCkqO.d.ts → engine-nB64f48I.d.ts} +18 -31
- package/dist/engine-nB64f48I.d.ts.map +1 -0
- package/dist/{eval-campaign-CfLQQs9B.js → eval-campaign-DNjCvAm-.js} +7 -6
- package/dist/{eval-campaign-CfLQQs9B.js.map → eval-campaign-DNjCvAm-.js.map} +1 -1
- package/dist/{exact-types-CBYF5MGd.d.ts → exact-types-Djvzosly.d.ts} +2 -2
- package/dist/{exact-types-CBYF5MGd.d.ts.map → exact-types-Djvzosly.d.ts.map} +1 -1
- package/dist/exec-BLtYZdWo.js +49 -0
- package/dist/exec-BLtYZdWo.js.map +1 -0
- package/dist/experiment/index.d.ts +802 -0
- package/dist/experiment/index.d.ts.map +1 -0
- package/dist/experiment/index.js +1108 -0
- package/dist/experiment/index.js.map +1 -0
- package/dist/experiment-tracker-CnRICnMl.js +500 -0
- package/dist/experiment-tracker-CnRICnMl.js.map +1 -0
- package/dist/experiment-tracker-IMntXr6J.d.ts +299 -0
- package/dist/experiment-tracker-IMntXr6J.d.ts.map +1 -0
- package/dist/{external-optimizer-contracts-Q9c0L3eY.d.ts → external-optimizer-contracts-CZuJNcT5.d.ts} +2 -2
- package/dist/{external-optimizer-contracts-Q9c0L3eY.d.ts.map → external-optimizer-contracts-CZuJNcT5.d.ts.map} +1 -1
- package/dist/{feedback-trajectory-GgoS0-MK.d.ts → feedback-trajectory-Rh280oXo.d.ts} +3 -3
- package/dist/{feedback-trajectory-GgoS0-MK.d.ts.map → feedback-trajectory-Rh280oXo.d.ts.map} +1 -1
- package/dist/fuzz.d.ts +1 -1
- package/dist/hosted/index.d.ts +3 -3
- package/dist/{index-BZUe-ODI.d.ts → index-BZ3-y4YL.d.ts} +3 -3
- package/dist/{index-BZUe-ODI.d.ts.map → index-BZ3-y4YL.d.ts.map} +1 -1
- package/dist/{index-DSC51roc.d.ts → index-C3ssXVLv.d.ts} +2 -2
- package/dist/{index-DSC51roc.d.ts.map → index-C3ssXVLv.d.ts.map} +1 -1
- package/dist/{index-4XwggC10.d.ts → index-C5HOo4ZF2.d.ts} +4 -4
- package/dist/index-C5HOo4ZF2.d.ts.map +1 -0
- package/dist/{index-B6-B0zTB.d.ts → index-CvXXlyz7.d.ts} +2 -2
- package/dist/{index-B6-B0zTB.d.ts.map → index-CvXXlyz7.d.ts.map} +1 -1
- package/dist/{index-Dx1kF3Ez.d.ts → index-CwDrUMe0.d.ts} +2 -2
- package/dist/{index-Dx1kF3Ez.d.ts.map → index-CwDrUMe0.d.ts.map} +1 -1
- package/dist/{index-BIL5vxxt.d.ts → index-Sh2I0DRc.d.ts} +11 -645
- package/dist/index-Sh2I0DRc.d.ts.map +1 -0
- package/dist/index.d.ts +214 -404
- package/dist/index.d.ts.map +1 -1
- package/dist/index.js +233 -649
- package/dist/index.js.map +1 -1
- package/dist/{insight-report-DqEsugpr.d.ts → insight-report-C6h6F_4L.d.ts} +3 -3
- package/dist/{insight-report-DqEsugpr.d.ts.map → insight-report-C6h6F_4L.d.ts.map} +1 -1
- package/dist/{integrity-CNGUaGBY.d.ts → integrity-BuqEKu-x.d.ts} +2 -2
- package/dist/{integrity-CNGUaGBY.d.ts.map → integrity-BuqEKu-x.d.ts.map} +1 -1
- package/dist/integrity-MLzHOfV9.js +141 -0
- package/dist/integrity-MLzHOfV9.js.map +1 -0
- package/dist/kind-factory-BHIgPmzS.js.map +1 -1
- package/dist/{llm-client-Dv5BiKLE.js → llm-client-DzvMUsS_.js} +24 -6
- package/dist/llm-client-DzvMUsS_.js.map +1 -0
- package/dist/matrix/index.d.ts +2 -2
- package/dist/meta-eval/index.d.ts +1 -1
- package/dist/{mint-DD-0oQTA.js → mint-CGEkzPLf.js} +2 -2
- package/dist/{mint-DD-0oQTA.js.map → mint-CGEkzPLf.js.map} +1 -1
- package/dist/{multi-layer-verifier-BHY1gWAc.d.ts → multi-layer-verifier-DnAqwl0h.d.ts} +2 -2
- package/dist/{multi-layer-verifier-BHY1gWAc.d.ts.map → multi-layer-verifier-DnAqwl0h.d.ts.map} +1 -1
- package/dist/multishot/index.d.ts +2 -2
- package/dist/openapi.json +1 -1
- package/dist/paired-promotion-decision-B6zJ3gYM.d.ts +114 -0
- package/dist/paired-promotion-decision-B6zJ3gYM.d.ts.map +1 -0
- package/dist/pipelines/index.d.ts +2 -1
- package/dist/pipelines/index.d.ts.map +1 -1
- package/dist/pre-registration-DakwTRXk.js +96 -0
- package/dist/pre-registration-DakwTRXk.js.map +1 -0
- package/dist/prime-bridge-transport-6feEglLf.d.ts +49 -0
- package/dist/prime-bridge-transport-6feEglLf.d.ts.map +1 -0
- package/dist/prime-protocol-BfSalTfR.js +453 -0
- package/dist/prime-protocol-BfSalTfR.js.map +1 -0
- package/dist/profile-cell.js +242 -1
- package/dist/profile-cell.js.map +1 -0
- package/dist/promotion-policy-ChWhTDBH.d.ts +289 -0
- package/dist/promotion-policy-ChWhTDBH.d.ts.map +1 -0
- package/dist/promotion-policy-CrLrmys8.js +682 -0
- package/dist/promotion-policy-CrLrmys8.js.map +1 -0
- package/dist/{propose-review-control-BciUCZoh.js → propose-review-control-qgWLA6E9.js} +2 -2
- package/dist/{propose-review-control-BciUCZoh.js.map → propose-review-control-qgWLA6E9.js.map} +1 -1
- package/dist/{release-report-ChOgpIoQ.d.ts → release-report-CI8uisI1.d.ts} +2 -2
- package/dist/{release-report-ChOgpIoQ.d.ts.map → release-report-CI8uisI1.d.ts.map} +1 -1
- package/dist/{release-report-Sl0xfkFv.js → release-report-Dy39cbFF.js} +3 -3
- package/dist/{release-report-Sl0xfkFv.js.map → release-report-Dy39cbFF.js.map} +1 -1
- package/dist/{replay-Krvb114g.d.ts → replay-DFf-teiC.d.ts} +5 -4
- package/dist/replay-DFf-teiC.d.ts.map +1 -0
- package/dist/{replay-GW61ezMW.js → replay-DyBLaKFc.js} +3 -3
- package/dist/{replay-GW61ezMW.js.map → replay-DyBLaKFc.js.map} +1 -1
- package/dist/reporting.d.ts +3 -3
- package/dist/reporting.js +2 -2
- package/dist/{researcher-xLeNcpKX.d.ts → researcher-BoaxeCzP.d.ts} +4 -4
- package/dist/{researcher-xLeNcpKX.d.ts.map → researcher-BoaxeCzP.d.ts.map} +1 -1
- package/dist/{reward-hacking-CyuzxKly.js → reward-hacking-B2tPL9a3.js} +12 -3
- package/dist/reward-hacking-B2tPL9a3.js.map +1 -0
- package/dist/{reward-hacking-RZgnGWlx.d.ts → reward-hacking-Cf1PtEOz.d.ts} +33 -3
- package/dist/reward-hacking-Cf1PtEOz.d.ts.map +1 -0
- package/dist/rl.d.ts +17 -7
- package/dist/rl.d.ts.map +1 -1
- package/dist/rl.js +16 -7
- package/dist/rl.js.map +1 -1
- package/dist/rollout/index.d.ts +2 -2
- package/dist/rollout/index.js +2 -2
- package/dist/{rollout-C2fD1cf4.js → rollout-2ECTXb2N.js} +2 -2
- package/dist/{rollout-C2fD1cf4.js.map → rollout-2ECTXb2N.js.map} +1 -1
- package/dist/{run-evidence-C6G41MSI.d.ts → run-evidence-BDFFai9R.d.ts} +2 -2
- package/dist/{run-evidence-C6G41MSI.d.ts.map → run-evidence-BDFFai9R.d.ts.map} +1 -1
- package/dist/{run-record-CWN8-VsV.js → run-record-DqOw5X6_.js} +2 -2
- package/dist/{run-record-CWN8-VsV.js.map → run-record-DqOw5X6_.js.map} +1 -1
- package/dist/{schema-Cef2cFmb.d.ts → schema-Cef2cFmb2.d.ts} +1 -1
- package/dist/schema-Cef2cFmb2.d.ts.map +1 -0
- package/dist/{semantic-concept-judge-DwF6n05O.js → semantic-concept-judge-Bmrq6yqU.js} +2 -2
- package/dist/{semantic-concept-judge-DwF6n05O.js.map → semantic-concept-judge-Bmrq6yqU.js.map} +1 -1
- package/dist/sequential-D-BLJBKU.js +299 -0
- package/dist/sequential-D-BLJBKU.js.map +1 -0
- package/dist/{server-D6XJQHw7.js → server-iu0ede49.js} +2 -2
- package/dist/{server-D6XJQHw7.js.map → server-iu0ede49.js.map} +1 -1
- package/dist/{single-run-lock-D5iN0Xzb.js → single-run-lock-BMQEv1wG.js} +7 -161
- package/dist/single-run-lock-BMQEv1wG.js.map +1 -0
- package/dist/{skill-usage-GlOphAhX.d.ts → skill-usage-CJlWEUFt.d.ts} +10 -10
- package/dist/{skill-usage-GlOphAhX.d.ts.map → skill-usage-CJlWEUFt.d.ts.map} +1 -1
- package/dist/{skillopt-optimization-method-7S43rbDB.d.ts → skillopt-optimization-method-BO7NAl3b.d.ts} +8 -294
- package/dist/skillopt-optimization-method-BO7NAl3b.d.ts.map +1 -0
- package/dist/{skillopt-optimization-method-Bfb-vBKe.js → skillopt-optimization-method-CQwZ-ZX8.js} +8 -652
- package/dist/skillopt-optimization-method-CQwZ-ZX8.js.map +1 -0
- package/dist/statistical-heldout-Dn9ruizm.d.ts +555 -0
- package/dist/statistical-heldout-Dn9ruizm.d.ts.map +1 -0
- package/dist/{statistics-C-dm-J6H.d.ts → statistics-D6Uebe_4.d.ts} +2 -2
- package/dist/{statistics-C-dm-J6H.d.ts.map → statistics-D6Uebe_4.d.ts.map} +1 -1
- package/dist/steps-BArUxhna.d.ts +51 -0
- package/dist/steps-BArUxhna.d.ts.map +1 -0
- package/dist/{integrity-fdt8XPAv.js → store-DNe_Uv1Q.js} +2 -140
- package/dist/store-DNe_Uv1Q.js.map +1 -0
- package/dist/{summary-report-B0cAyA7N.d.ts → summary-report-DuUS_i7W.d.ts} +3 -114
- package/dist/summary-report-DuUS_i7W.d.ts.map +1 -0
- package/dist/{summary-report-9A5y7EsK.js → summary-report-Lf-5I7xh.js} +2 -2
- package/dist/{summary-report-9A5y7EsK.js.map → summary-report-Lf-5I7xh.js.map} +1 -1
- package/dist/supervisor-run/index.d.ts +2 -2
- package/dist/supervisor-run/index.js +1 -1
- package/dist/{supervisor-run-DiyQVczd.js → supervisor-run-D_sokXcO.js} +32 -8
- package/dist/{supervisor-run-DiyQVczd.js.map → supervisor-run-D_sokXcO.js.map} +1 -1
- package/dist/tool-groups-DIVBnJyl.d.ts +28 -0
- package/dist/tool-groups-DIVBnJyl.d.ts.map +1 -0
- package/dist/trace-repair/index.d.ts +2102 -0
- package/dist/trace-repair/index.d.ts.map +1 -0
- package/dist/trace-repair/index.js +3878 -0
- package/dist/trace-repair/index.js.map +1 -0
- package/dist/traces.d.ts +6 -6
- package/dist/traces.js +3 -2
- package/dist/trajectory-YC15QDYQ.d.ts +24 -0
- package/dist/trajectory-YC15QDYQ.d.ts.map +1 -0
- package/dist/trajectory-replay/index.d.ts +781 -0
- package/dist/trajectory-replay/index.d.ts.map +1 -0
- package/dist/trajectory-replay/index.js +2103 -0
- package/dist/trajectory-replay/index.js.map +1 -0
- package/dist/{types-XMVEdrE_.d.ts → types-D216SgwM.d.ts} +24 -6
- package/dist/types-D216SgwM.d.ts.map +1 -0
- package/dist/{types-yLK8gXE9.d.ts → types-D4mog56g.d.ts} +2 -2
- package/dist/{types-yLK8gXE9.d.ts.map → types-D4mog56g.d.ts.map} +1 -1
- package/dist/{types-BhP9q0Fq.d.ts → types-DF_Udrp-.d.ts} +52 -3
- package/dist/{types-BhP9q0Fq.d.ts.map → types-DF_Udrp-.d.ts.map} +1 -1
- package/dist/{types-DOZyvsFU.d.ts → types-DYuNHo9R.d.ts} +3 -3
- package/dist/{types-DOZyvsFU.d.ts.map → types-DYuNHo9R.d.ts.map} +1 -1
- package/dist/usage-receipt-EVI8B8Xu.js.map +1 -1
- package/dist/verdict-DExhxfgR.d.ts +201 -0
- package/dist/verdict-DExhxfgR.d.ts.map +1 -0
- package/dist/verdict-cache-BCcOh0kF.js +159 -0
- package/dist/verdict-cache-BCcOh0kF.js.map +1 -0
- package/dist/wire/index.d.ts +2 -2
- package/dist/wire/index.js +1 -1
- package/docs/charter.md +112 -0
- package/docs/experiment.md +104 -0
- package/docs/prime-analyst.md +1 -0
- package/docs/trace-analysis.md +26 -0
- package/docs/trace-repair-admission.md +194 -0
- package/docs/trace-repair-analyst-arms.md +121 -0
- package/docs/trace-repair-continuation.md +107 -0
- package/docs/trace-repair-grader.md +163 -0
- package/docs/trajectory-replay.md +110 -0
- package/docs/verification-strategies.md +103 -0
- package/package.json +19 -2
- package/dist/agent-profile-cell-CbfBm2g6.js +0 -335
- package/dist/agent-profile-cell-CbfBm2g6.js.map +0 -1
- package/dist/baseline-D_fT6277.d.ts.map +0 -1
- package/dist/benchmark-J9Qe6j2_.d.ts.map +0 -1
- package/dist/benchmark-command-CQd78YHt.js.map +0 -1
- package/dist/campaign-CXsdyym7.js.map +0 -1
- package/dist/default-registry-J9m-_tya.d.ts.map +0 -1
- package/dist/index-4XwggC10.d.ts.map +0 -1
- package/dist/index-BIL5vxxt.d.ts.map +0 -1
- package/dist/integrity-fdt8XPAv.js.map +0 -1
- package/dist/llm-client-Dv5BiKLE.js.map +0 -1
- package/dist/replay-Krvb114g.d.ts.map +0 -1
- package/dist/reward-hacking-CyuzxKly.js.map +0 -1
- package/dist/reward-hacking-RZgnGWlx.d.ts.map +0 -1
- package/dist/schema-Cef2cFmb.d.ts.map +0 -1
- package/dist/single-run-lock-D5iN0Xzb.js.map +0 -1
- package/dist/skillopt-optimization-method-7S43rbDB.d.ts.map +0 -1
- package/dist/skillopt-optimization-method-Bfb-vBKe.js.map +0 -1
- package/dist/summary-report-B0cAyA7N.d.ts.map +0 -1
- package/dist/tool-groups-CK0JCkqO.d.ts.map +0 -1
- package/dist/types-XMVEdrE_.d.ts.map +0 -1
- package/dist/verdict-Dps8_okt.d.ts +0 -37
- package/dist/verdict-Dps8_okt.d.ts.map +0 -1
package/dist/index.d.ts
CHANGED
|
@@ -1,44 +1,51 @@
|
|
|
1
1
|
import { a as JudgeError, c as ReplayError, i as ConfigError, l as ValidationError, n as AgentEvalErrorCode, o as LimitExceededError, r as CaptureIntegrityError, s as NotFoundError, t as AgentEvalError, u as VerificationError } from "./errors-CKPfb2aH.js";
|
|
2
2
|
import { C as verifyAgentProfileCell, S as validateAgentProfileCell, _ as buildAgentInterfaceProfileCell, a as AgentProfileCellSchemaVersion, b as requireAgentProfileCell, c as AgentProfileHarness, d as AgentProfileKind, f as AgentProfileSource, g as assertRunAgentProfileCell, h as agentProfileCellKey, i as AgentProfileCellInput, l as AgentProfileJson, m as agentProfileCellHashMaterial, n as AgentInterfaceProfileLike, o as AgentProfileCellValidationError, p as AgentProfileSourceInput, r as AgentProfileCell, s as AgentProfileDimensionValue, t as AGENT_PROFILE_KINDS, u as AgentProfileJsonObject, v as buildAgentProfileCell, x as toAgentProfileJson, y as groupRunsByAgentProfileCell } from "./agent-profile-cell-BOP-iA9Q.js";
|
|
3
|
-
import { t as DefaultVerdict } from "./verdict-
|
|
4
|
-
import { a as MultiLayerVerifier, c as VerifyContext, i as LayerStatus, l as VerifyOptions, n as Layer, o as Severity, r as LayerResult, s as VerificationReport, t as Finding, u as gradeSemanticStatus } from "./multi-layer-verifier-
|
|
5
|
-
import { $ as RunScore, B as ConceptSpec, E as FindingSubjectKind, G as SemanticConceptJudgeOptions, H as DEFAULT_COMPLEXITY_WEIGHTS, J as runSemanticConceptJudge, K as SemanticConceptJudgeResult, L as defineTraceAnalyst, M as DspyRlmTraceEngineOptions, N as createDspyRlmTraceEngine, Q as DEFAULT_RUN_SCORE_WEIGHTS, R as ConceptComplexity, T as FindingSubject, U as SEMANTIC_CONCEPT_JUDGE_VERSION, V as ConceptWeightStrategy, W as SemanticConceptJudgeInput, X as RunCriticOptions, Y as RunCritic, Z as RunTrace, _ as FindingsDiff, b as defaultIsMaterial, c as DEFAULT_TRACE_ANALYST_KINDS, d as IMPROVEMENT_KIND_SPEC, et as RunScoreWeights, f as FAILURE_MODE_KIND_SPEC, g as DiffPolicy, h as emitControlIntegrityFindings, l as KNOWLEDGE_POISONING_KIND_SPEC, m as ControlIntegrityAnalyst, n as SkillUsageAnalyst, nt as clamp01, p as CONTROL_INTEGRITY_ANALYST, q as createSemanticConceptJudge, t as SKILL_USAGE_ANALYST, tt as aggregateRunScore, u as KNOWLEDGE_GAP_KIND_SPEC, v as FindingsStore, x as diffFindings, y as PersistedFinding, z as ConceptFinding } from "./skill-usage-
|
|
3
|
+
import { a as StrategyChecker, c as VerificationStrategyProfile, i as CheckerOutcome, l as VerificationStrategySource, n as VerdictCertification, o as VERIFICATION_STRATEGIES, r as CheckerIdentity, s as VERIFICATION_STRATEGY_SOURCES, t as DefaultVerdict } from "./verdict-DExhxfgR.js";
|
|
4
|
+
import { a as MultiLayerVerifier, c as VerifyContext, i as LayerStatus, l as VerifyOptions, n as Layer, o as Severity, r as LayerResult, s as VerificationReport, t as Finding, u as gradeSemanticStatus } from "./multi-layer-verifier-DnAqwl0h.js";
|
|
5
|
+
import { $ as RunScore, B as ConceptSpec, E as FindingSubjectKind, G as SemanticConceptJudgeOptions, H as DEFAULT_COMPLEXITY_WEIGHTS, J as runSemanticConceptJudge, K as SemanticConceptJudgeResult, L as defineTraceAnalyst, M as DspyRlmTraceEngineOptions, N as createDspyRlmTraceEngine, Q as DEFAULT_RUN_SCORE_WEIGHTS, R as ConceptComplexity, T as FindingSubject, U as SEMANTIC_CONCEPT_JUDGE_VERSION, V as ConceptWeightStrategy, W as SemanticConceptJudgeInput, X as RunCriticOptions, Y as RunCritic, Z as RunTrace, _ as FindingsDiff, b as defaultIsMaterial, c as DEFAULT_TRACE_ANALYST_KINDS, d as IMPROVEMENT_KIND_SPEC, et as RunScoreWeights, f as FAILURE_MODE_KIND_SPEC, g as DiffPolicy, h as emitControlIntegrityFindings, l as KNOWLEDGE_POISONING_KIND_SPEC, m as ControlIntegrityAnalyst, n as SkillUsageAnalyst, nt as clamp01, p as CONTROL_INTEGRITY_ANALYST, q as createSemanticConceptJudge, t as SKILL_USAGE_ANALYST, tt as aggregateRunScore, u as KNOWLEDGE_GAP_KIND_SPEC, v as FindingsStore, x as diffFindings, y as PersistedFinding, z as ConceptFinding } from "./skill-usage-CJlWEUFt.js";
|
|
6
6
|
import { C as PendingCostCall, D as costForUsage, E as costForTokenPricing, O as modelPriceKey, S as PaidCallResult, T as RunPaidCallInput, _ as CostReservationExceededError, a as CostChannel, b as CustomTokenPricing, c as CostLedgerHandle, d as CostLedgerPersistenceError, f as CostLedgerSummary, g as CostReceiptInput, h as CostReceiptCaptureError, i as CostCeilingReachedError, l as CostLedgerOptions, m as CostReceipt, n as CostAccountingIncompleteError, o as CostLedger, p as CostProvenance, r as CostCallConflictError, s as CostLedgerFilter, t as ChannelRollup, u as CostLedgerPersistence, v as CostResult, w as PendingCostCallView, x as MaximumCharge, y as CostUsage } from "./cost-ledger-Bv_e8XHY.js";
|
|
7
7
|
import { C as TraceEvent, D as isSandboxSpan, E as isRetrievalSpan, O as isToolSpan, S as ToolSpan, T as isLlmSpan, _ as Span, a as FAILURE_CLASSES, b as SpanStatus, c as JudgeSpan, d as RetrievalSpan, f as Run, g as SandboxSpan, h as RunStatus, i as EventKind, l as LlmSpan, n as BudgetLedgerEntry, o as FailureClass, p as RunLayer, r as BudgetSpec, s as GenericSpan, t as Artifact, u as Message, v as SpanBase, w as isJudgeSpan, x as TRACE_SCHEMA_VERSION, y as SpanKind } from "./schema-BtVldJ3T.js";
|
|
8
8
|
import { a as RunRecord, c as RunTaskFailure, d as isRunRecord, f as modelHasSnapshot, g as validateRunRecord, h as runTaskScore, i as RunOutcome, l as RunTerminalOutcome, m as roundTripRunRecord, n as RunCostProvenance, o as RunRecordValidationError, p as parseRunRecordSafe, r as RunJudgeMetadata, s as RunSplitTag, t as JudgeScoresRecord, u as RunTokenUsage } from "./run-record-DdSa93_W.js";
|
|
9
|
-
import { a as ExtractedUsage, c as extractUsageFromResponse, i as ExtractUsageFromSseOptions, l as extractUsageFromSse, n as CaptureFetchOptions, o as SseUsageMode, r as captureFetchToRawSink, s as extractUsage, t as CaptureFetchContext } from "./index-
|
|
10
|
-
import { $ as assertLlmRoute, A as ChatClient, At as
|
|
9
|
+
import { a as ExtractedUsage, c as extractUsageFromResponse, i as ExtractUsageFromSseOptions, l as extractUsageFromSse, n as CaptureFetchOptions, o as SseUsageMode, r as captureFetchToRawSink, s as extractUsage, t as CaptureFetchContext } from "./index-CwDrUMe0.js";
|
|
10
|
+
import { $ as assertLlmRoute, A as ChatClient, At as InMemoryRawProviderSinkOptions, B as SandboxSdkTransportOpts, C as ScenarioFile, Ct as CrossFamilyError, D as TurnMetrics, Dt as FileSystemRawProviderSink, E as Turn, Et as judgeFamily, F as CreateChatClientOpts, Ft as RawProviderSink, G as LlmCallResult, H as LlmCallError, I as CustomTransportOpts, It as RawProviderSinkFilter, J as LlmMessage, K as LlmClient, L as DirectProviderTransportOpts, Lt as defaultProviderRedactor, M as ChatResponse, Mt as ProviderRedactor, N as ChatTransport, Nt as RawProviderDirection, O as TurnResult, Ot as FileSystemRawProviderSinkOptions, P as CliBridgeTransportOpts, Pt as RawProviderEvent, Q as LlmUsage, R as MockTransportOpts, Rt as providerFromBaseUrl, S as Scenario, St as AssertCrossFamilyOptions, T as TestResult, Tt as assertCrossFamily, U as LlmCallMetadata, V as createChatClient, W as LlmCallRequest, X as LlmRouteAssertionError, Y as LlmResponseError, Z as LlmRouteRequirements, _ as PersonaConfig, _t as assertServedModel, a as CheckResult, at as isTransientLlmError, b as RouteMap, bt as normalizeModelId, c as DriverResult, ct as stripFencedJson, d as FeedbackPattern, dt as ModelSubstitutionError, et as backoffMs, f as JudgeConfig, ft as PROBE_MAX_TOKENS, g as JudgeScore, gt as assertCrossFamilyServed, h as JudgeRubric, ht as ServedModelVerdict, i as BenchmarkRunnerConfig, it as costReceiptFromLlmError, j as ChatRequest, jt as NoopRawProviderSink, k as ChatCallOpts, kt as InMemoryRawProviderSink, l as DriverState, lt as AssertCrossFamilyServedOptions, m as JudgeInput, mt as ServedModelCheck, n as ArtifactResult, nt as callLlmJson, o as CollectedArtifacts, ot as maximumChargeForLlmRequest, p as JudgeFn, pt as ServedCrossFamilyError, q as LlmClientOptions, r as BenchmarkReport, rt as costReceiptFromLlm, s as CompletionCriterion, st as probeLlm, t as ArtifactCheck, tt as callLlm, u as EvalResult, ut as AssertServedModelOptions, v as PersonaRigor, vt as assertServedModels, w as ScenarioResult, wt as JudgeFamily, x as RubricDimension, xt as servedModelAcceptable, y as ProductClientConfig, yt as checkServedModel, z as RouterTransportOpts } from "./types-D216SgwM.js";
|
|
11
11
|
import { a as RunFilter, i as InMemoryTraceStore, n as FileSystemTraceStore, o as SpanFilter, r as FileSystemTraceStoreOptions, s as TraceStore, t as EventFilter } from "./store-CT9YIIve.js";
|
|
12
12
|
import { a as TraceEmitterOptions, i as TraceEmitter, n as RunCompleteHookContext, o as llmSpanFromProvider, r as SpanHandle, t as RunCompleteHook } from "./emitter-DGQGoLyj.js";
|
|
13
|
-
import { a as RunIntegrityReport, i as RunIntegrityIssueCode, n as RunIntegrityExpectations, o as assertRunCaptured, r as RunIntegrityIssue, s as throwIfRunIncomplete, t as RunIntegrityError } from "./integrity-
|
|
14
|
-
import { $ as traceAnalystOnRunComplete, A as stringField, At as OtlpSpanRoleInput, B as TraceInsightReadiness, Bt as exportRunAsOtlp, C as ProjectedOtlpSpan, Ct as createOtelTracingStore, D as inferOtlpKind, Dt as OtelExporter, E as firstStringAttr, Et as OtelExportConfig, F as TraceInsightFinding, Ft as traceSpanKindToOpenInferenceKind, G as defaultTraceInsightPanel, H as TraceInsightTask, I as TraceInsightPanelRole, It as OTEL_AGENT_EVAL_SCOPE, J as inferDomainKeywords, K as describeTraceInsightScope, L as TraceInsightPromptInput, Lt as OtlpExport, M as flattenOtlpExportToNdjson, Mt as applyToolSpanOtlpAttributes, N as OtlpFlatLine, Nt as classifyOtlpSpanRole, O as projectOtlpFlatLine, Ot as createOtelExporter, P as TraceInsightContext, Pt as isOtlpModelCall, Q as TraceAnalystHookOptions, R as TraceInsightQualityGate, Rt as OtlpResourceSpans, S as otlpToTraceRunRecords, St as redactValue, T as extractOtlpAttributes, Tt as ExportableSpan, U as buildTraceInsightContext, V as TraceInsightSuite, W as buildTraceInsightPrompt, X as scoreTraceInsightReadiness, Y as planTraceInsightQuestions, Z as tokenizeDomainWords, _ as OtlpTraceRunRecord, _t as DEFAULT_REDACTION_RULES, a as ReplayFetchOptions, at as TraceFileMissingError, b as otlpRowsToTraceRunRecords, bt as RedactionRule, c as ToolTraceMissingError, ct as AnalyzeTracesInput, d as OtlpFileTraceStoreOptions, dt as analyzeTraces, et as SpanNotFoundError, f as ToolSpansToTraceAnalysisStoreOptions, ft as createBoundedTraceAnalysisStore, g as OtlpToRunRecordsOptions, gt as convertTraceStoresToOtlp, h as TRACE_ANALYST_ACTOR_DESCRIPTION_VERSION, ht as TracesToOtlpResult, i as ReplayCacheStats, it as TraceFileMalformedError, j as FlattenOtlpOptions, jt as ToolSpanOtlpInput, k as readOtlpStatus, kt as OtlpSpanRole, l as toolSpansToTraceAnalysisStore, lt as AnalyzeTracesOptions, m as TRACE_ANALYST_ACTOR_DESCRIPTION, mt as TraceStoreToOtlpOptions, n as ReplayCacheEntry, nt as TraceAnalysisStoreContractError, o as createReplayFetch, ot as TraceFileTooLargeError, p as otlpTextToTraceAnalysisStore, pt as TraceStoreSource, q as domainEvidencePattern, r as ReplayCacheMissError, rt as TraceAnalysisValidationError, s as iterateRawCalls, st as TraceNotFoundError, t as ReplayCache, tt as TraceAnalysisLimitError, u as OtlpFileTraceStore, ut as AnalyzeTracesResult, v as TraceAggregate, vt as REDACTION_VERSION, w as asString, wt as otelRunCompleteHook, x as otlpToRunRecords, xt as redactString, y as otlpRowsToRunRecords, yt as RedactionReport, z as TraceInsightQuestion, zt as OtlpSpan } from "./replay-
|
|
13
|
+
import { a as RunIntegrityReport, i as RunIntegrityIssueCode, n as RunIntegrityExpectations, o as assertRunCaptured, r as RunIntegrityIssue, s as throwIfRunIncomplete, t as RunIntegrityError } from "./integrity-BuqEKu-x.js";
|
|
14
|
+
import { $ as traceAnalystOnRunComplete, A as stringField, At as OtlpSpanRoleInput, B as TraceInsightReadiness, Bt as exportRunAsOtlp, C as ProjectedOtlpSpan, Ct as createOtelTracingStore, D as inferOtlpKind, Dt as OtelExporter, E as firstStringAttr, Et as OtelExportConfig, F as TraceInsightFinding, Ft as traceSpanKindToOpenInferenceKind, G as defaultTraceInsightPanel, H as TraceInsightTask, I as TraceInsightPanelRole, It as OTEL_AGENT_EVAL_SCOPE, J as inferDomainKeywords, K as describeTraceInsightScope, L as TraceInsightPromptInput, Lt as OtlpExport, M as flattenOtlpExportToNdjson, Mt as applyToolSpanOtlpAttributes, N as OtlpFlatLine, Nt as classifyOtlpSpanRole, O as projectOtlpFlatLine, Ot as createOtelExporter, P as TraceInsightContext, Pt as isOtlpModelCall, Q as TraceAnalystHookOptions, R as TraceInsightQualityGate, Rt as OtlpResourceSpans, S as otlpToTraceRunRecords, St as redactValue, T as extractOtlpAttributes, Tt as ExportableSpan, U as buildTraceInsightContext, V as TraceInsightSuite, W as buildTraceInsightPrompt, X as scoreTraceInsightReadiness, Y as planTraceInsightQuestions, Z as tokenizeDomainWords, _ as OtlpTraceRunRecord, _t as DEFAULT_REDACTION_RULES, a as ReplayFetchOptions, at as TraceFileMissingError, b as otlpRowsToTraceRunRecords, bt as RedactionRule, c as ToolTraceMissingError, ct as AnalyzeTracesInput, d as OtlpFileTraceStoreOptions, dt as analyzeTraces, et as SpanNotFoundError, f as ToolSpansToTraceAnalysisStoreOptions, ft as createBoundedTraceAnalysisStore, g as OtlpToRunRecordsOptions, gt as convertTraceStoresToOtlp, h as TRACE_ANALYST_ACTOR_DESCRIPTION_VERSION, ht as TracesToOtlpResult, i as ReplayCacheStats, it as TraceFileMalformedError, j as FlattenOtlpOptions, jt as ToolSpanOtlpInput, k as readOtlpStatus, kt as OtlpSpanRole, l as toolSpansToTraceAnalysisStore, lt as AnalyzeTracesOptions, m as TRACE_ANALYST_ACTOR_DESCRIPTION, mt as TraceStoreToOtlpOptions, n as ReplayCacheEntry, nt as TraceAnalysisStoreContractError, o as createReplayFetch, ot as TraceFileTooLargeError, p as otlpTextToTraceAnalysisStore, pt as TraceStoreSource, q as domainEvidencePattern, r as ReplayCacheMissError, rt as TraceAnalysisValidationError, s as iterateRawCalls, st as TraceNotFoundError, t as ReplayCache, tt as TraceAnalysisLimitError, u as OtlpFileTraceStore, ut as AnalyzeTracesResult, v as TraceAggregate, vt as REDACTION_VERSION, w as asString, wt as otelRunCompleteHook, x as otlpToRunRecords, xt as redactString, y as otlpRowsToRunRecords, yt as RedactionReport, z as TraceInsightQuestion, zt as OtlpSpan } from "./replay-DFf-teiC.js";
|
|
15
15
|
import { C as TOOL_LATENCY_MS, D as asNumber, E as applyLlmSpanOtlpAttributes, O as contextInputTokens, S as TOOL_ARGS_CAPTURED, T as TOOL_NAME_ATTR_KEYS, _ as LlmSpanOtlpInput, a as LLM_CACHE_WRITE_TOKEN_ATTR_KEYS, b as RUN_COST_ATTR_KEYS, c as LLM_COST_USD, d as LLM_MODEL_ATTR_KEYS, f as LLM_MODEL_NAME, g as LLM_REASONING_TOKEN_ATTR_KEYS, h as LLM_REASONING_TOKENS, i as LLM_CACHE_WRITE_TOKENS, k as firstNumberAttr, l as LLM_INPUT_TOKENS, m as LLM_OUTPUT_TOKEN_ATTR_KEYS, n as LLM_CACHED_TOKENS, o as LLM_CONTEXT_TOKENS, p as LLM_OUTPUT_TOKENS, r as LLM_CACHED_TOKEN_ATTR_KEYS, s as LLM_COST_ATTR_KEYS, t as INPUT_VALUE, u as LLM_INPUT_TOKEN_ATTR_KEYS, v as OPENINFERENCE_SPAN_KIND, w as TOOL_NAME, x as SPAN_KIND_ATTR_KEYS, y as OUTPUT_VALUE } from "./attribute-vocabulary-DLJ6303h.js";
|
|
16
16
|
import { a as judgeSpans, c as runsForScenario, i as hasCapturedToolArgs, l as toolSpans, n as argHash, o as llmSpans, r as groupBy, s as runFailureClass, t as aggregateLlm } from "./query-CJ_DX8vl.js";
|
|
17
|
-
import { A as
|
|
18
|
-
import { a as createTraceAnalyst, c as runTraceAnalyst, f as BehavioralMetrics, g as computeTraceMetrics, h as SuboptimalSignal, i as TraceAnalystDefinition, m as SuboptimalCode, n as buildDefaultAnalystRegistry, o as renderPriorFindings, p as BehavioralTokenSequence, r as CreateTraceAnalystOptions, s as renderUpstreamFindings, t as DefaultAnalystRegistryOptions } from "./default-registry-
|
|
19
|
-
import { a as ExactAnalystRunPolicySnapshot, c as ExactAnalystSnapshot, d as ExactExecutionComponentSnapshot, i as ExactAnalystRunEvent, l as ExactCapableAnalyst, n as ExactAnalystExecutionPlanSnapshot, o as ExactAnalystRunResult, r as ExactAnalystRunCompletion, s as ExactAnalystRunSummary, t as ExactAnalystBudgetSnapshot, u as ExactExecutionComponentIdentity } from "./exact-types-
|
|
20
|
-
import { A as ExactAnalystRunExecutionError, C as jsonHasKeys, D as AnalystRegistryOptions, E as AnalystRegistry, M as RegistryRunOpts, N as assertExactRegistryRunOpts, O as BudgetPolicy, S as containsAll, T as AnalystHooks, _ as ValidationContext, a as ProducedProposal, b as byteLengthRange, c as SatisfiedBy, d as createLlmCorrectnessChecker, f as createTokenRecallChecker, g as ArtifactValidator, h as Artifact$1, i as LlmCorrectnessCheckerOpts, j as ExactRegistryRunOpts, k as ExactAnalystBudgetPolicy, l as TaskGold, m as verifyCompletion, n as CompletionVerdict, o as ProducedState, p as parseCorrectnessResponse, r as CorrectnessChecker, s as RequirementCheck, t as CompletionRequirement, u as completionVerdict, v as ValidationIssue, w as regexMatch, x as composeValidators, y as ValidationResult } from "./completion-verifier-
|
|
21
|
-
import {
|
|
17
|
+
import { A as SearchSpanResult, B as ViewSpansResult, C as TRACE_ANALYSIS_LIMITS, D as DatasetOverview, E as DEFAULT_TRACE_ANALYST_BUDGETS, F as TraceAnalystFilters, H as ViewTraceResult, I as TraceAnalystSpan, L as TraceAnalystSpanKind, M as SpanMatchRecord, N as TRACE_ANALYST_TRUNCATION_MARKER_PREFIX, O as ErrorCluster, P as TraceAnalystByteBudgets, R as TraceAnalystSpanStatus, S as BoundedTraceAnalysisStoreOptions, T as TraceAnalysisStoreContext, V as ViewTraceOversized, _ as ProposalFinding, a as AnalystInputKind, b as makeFinding, c as AnalystRunInputs, d as AnalystSeverity, f as AnalystUsageReceipt, g as ExecutionProbeRequest, h as ExecutionProbeOutcome, i as AnalystFinding, j as SearchTraceResult, k as QueryTracesPage, l as AnalystRunResult, m as ExecutionProbe, n as AnalystContext, o as AnalystRequirements, p as EvidenceRef, r as AnalystCost, s as AnalystRunEvent, t as Analyst, u as AnalystRunSummary, v as ProposalFindingOrigin, w as TraceAnalysisStore, x as makeProposalFinding, y as computeFindingId, z as TraceAnalystTraceSummary } from "./types-DF_Udrp-.js";
|
|
18
|
+
import { a as createTraceAnalyst, c as runTraceAnalyst, f as BehavioralMetrics, g as computeTraceMetrics, h as SuboptimalSignal, i as TraceAnalystDefinition, m as SuboptimalCode, n as buildDefaultAnalystRegistry, o as renderPriorFindings, p as BehavioralTokenSequence, r as CreateTraceAnalystOptions, s as renderUpstreamFindings, t as DefaultAnalystRegistryOptions } from "./default-registry-BZhStdjl.js";
|
|
19
|
+
import { a as ExactAnalystRunPolicySnapshot, c as ExactAnalystSnapshot, d as ExactExecutionComponentSnapshot, i as ExactAnalystRunEvent, l as ExactCapableAnalyst, n as ExactAnalystExecutionPlanSnapshot, o as ExactAnalystRunResult, r as ExactAnalystRunCompletion, s as ExactAnalystRunSummary, t as ExactAnalystBudgetSnapshot, u as ExactExecutionComponentIdentity } from "./exact-types-Djvzosly.js";
|
|
20
|
+
import { A as ExactAnalystRunExecutionError, C as jsonHasKeys, D as AnalystRegistryOptions, E as AnalystRegistry, M as RegistryRunOpts, N as assertExactRegistryRunOpts, O as BudgetPolicy, S as containsAll, T as AnalystHooks, _ as ValidationContext, a as ProducedProposal, b as byteLengthRange, c as SatisfiedBy, d as createLlmCorrectnessChecker, f as createTokenRecallChecker, g as ArtifactValidator, h as Artifact$1, i as LlmCorrectnessCheckerOpts, j as ExactRegistryRunOpts, k as ExactAnalystBudgetPolicy, l as TaskGold, m as verifyCompletion, n as CompletionVerdict, o as ProducedState, p as parseCorrectnessResponse, r as CorrectnessChecker, s as RequirementCheck, t as CompletionRequirement, u as completionVerdict, v as ValidationIssue, w as regexMatch, x as composeValidators, y as ValidationResult } from "./completion-verifier-foUCLif_.js";
|
|
21
|
+
import { a as ProfileAxisSpec, c as agentProfileModelId, i as HarnessType, l as expandProfileAxes, n as CODING_HARNESSES, o as agentProfileHash, r as HARNESS_NATIVE_MODEL, s as agentProfileId, t as AgentProfile, u as harnessAxisOf } from "./agent-profile-CgDTo40f.js";
|
|
22
|
+
import { R as Scenario$1, S as JudgeConfig$1, w as JudgeScore$1 } from "./types-DYuNHo9R.js";
|
|
22
23
|
import { a as DatasetScenario, c as SliceOptions, i as DatasetProvenance, l as hashScenarios, n as DatasetDifficulty, o as DatasetSplit, r as DatasetManifest, s as HoldoutLockedError, t as Dataset } from "./dataset-C8xaLXdY.js";
|
|
23
|
-
import { $ as pairedCohensDz, A as WeightedCompositeInput, At as positionalBias, B as eProcess, C as RankTestMethod, Ct as GoldenItem, D as ScoreRiskDifferenceResult, Dt as calibrateJudge, E as RiskDifferenceResult, Et as VerbosityBiasResult, F as cliffsDelta, G as mannWhitneyU, H as interRaterReliability, I as cohensD, J as mcnemarRequiredN, K as mcnemar, L as confidenceInterval, M as WilcoxonSignedRankResult, Mt as verbosityBias, N as benjaminiHochberg, O as SignTestAlternative, Ot as calibrateJudgeContinuous, P as bonferroni, Q as pairedBootstrap, R as corpusInterRaterAgreement, S as ProportionInterval, St as ContinuousCalibrationResult, T as RankTestOptions, Tt as SelfPreferenceResult, U as interpretCliffs, V as holm, W as isBinaryOutcomeVector, X as normalizeScores, Y as mulberry32, Z as pairedBinaryScale, _ as McNemarResult, _t as wilson, a as CorpusAgreementReport, at as pairedSignTest, b as PairedSignTestResult, bt as ContinuousAgreement, c as DEFAULT_PERMUTATIONS, ct as passAtK, d as EProcessState, dt as requiredPairedSampleSize, et as pairedDeltaTieFraction, f as EProcessStep, ft as requiredSampleSize, g as MannWhitneyResult, gt as wilcoxonSignedRank, h as MANN_WHITNEY_EXACT_MAX_WORK, ht as weightedMean, i as CorpusAgreementPerDimension, it as pairedRiskDifferenceScore, j as WeightedCompositeResult, jt as selfPreference, k as WILCOXON_EXACT_MAX_N, kt as continuousAgreement, l as EProcess, lt as pearsonR, m as MANN_WHITNEY_EXACT_MAX_STATES, mt as weightedComposite, n as CliffsMagnitude, nt as pairedRiskDifference, o as CorpusScoreRecord, ot as pairedTTest, p as ExactRiskDifferenceResult, pt as spearmanR, q as mcnemarPower, r as CorpusAgreementOptions, rt as pairedRiskDifferenceExact, s as DECISION_PAIRED_DELTA_STATISTIC, st as partialCredit, t as BOOTSTRAP_GATE_MIN_N, tt as pairedMde, u as EProcessOptions, ut as ranks, v as PairedBootstrapOptions, vt as CalibrationResult, w as RankTestMethodRequest, wt as PositionalBiasResult, x as PairedTTestResult, xt as ContinuousAgreementOptions, y as PairedBootstrapResult, yt as CandidateScore, z as corpusInterRaterAgreementFromJudgeScores } from "./statistics-
|
|
24
|
-
import {
|
|
24
|
+
import { $ as pairedCohensDz, A as WeightedCompositeInput, At as positionalBias, B as eProcess, C as RankTestMethod, Ct as GoldenItem, D as ScoreRiskDifferenceResult, Dt as calibrateJudge, E as RiskDifferenceResult, Et as VerbosityBiasResult, F as cliffsDelta, G as mannWhitneyU, H as interRaterReliability, I as cohensD, J as mcnemarRequiredN, K as mcnemar, L as confidenceInterval, M as WilcoxonSignedRankResult, Mt as verbosityBias, N as benjaminiHochberg, O as SignTestAlternative, Ot as calibrateJudgeContinuous, P as bonferroni, Q as pairedBootstrap, R as corpusInterRaterAgreement, S as ProportionInterval, St as ContinuousCalibrationResult, T as RankTestOptions, Tt as SelfPreferenceResult, U as interpretCliffs, V as holm, W as isBinaryOutcomeVector, X as normalizeScores, Y as mulberry32, Z as pairedBinaryScale, _ as McNemarResult, _t as wilson, a as CorpusAgreementReport, at as pairedSignTest, b as PairedSignTestResult, bt as ContinuousAgreement, c as DEFAULT_PERMUTATIONS, ct as passAtK, d as EProcessState, dt as requiredPairedSampleSize, et as pairedDeltaTieFraction, f as EProcessStep, ft as requiredSampleSize, g as MannWhitneyResult, gt as wilcoxonSignedRank, h as MANN_WHITNEY_EXACT_MAX_WORK, ht as weightedMean, i as CorpusAgreementPerDimension, it as pairedRiskDifferenceScore, j as WeightedCompositeResult, jt as selfPreference, k as WILCOXON_EXACT_MAX_N, kt as continuousAgreement, l as EProcess, lt as pearsonR, m as MANN_WHITNEY_EXACT_MAX_STATES, mt as weightedComposite, n as CliffsMagnitude, nt as pairedRiskDifference, o as CorpusScoreRecord, ot as pairedTTest, p as ExactRiskDifferenceResult, pt as spearmanR, q as mcnemarPower, r as CorpusAgreementOptions, rt as pairedRiskDifferenceExact, s as DECISION_PAIRED_DELTA_STATISTIC, st as partialCredit, t as BOOTSTRAP_GATE_MIN_N, tt as pairedMde, u as EProcessOptions, ut as ranks, v as PairedBootstrapOptions, vt as CalibrationResult, w as RankTestMethodRequest, wt as PositionalBiasResult, x as PairedTTestResult, xt as ContinuousAgreementOptions, y as PairedBootstrapResult, yt as CandidateScore, z as corpusInterRaterAgreementFromJudgeScores } from "./statistics-D6Uebe_4.js";
|
|
25
|
+
import { a as PairedPromotionDecision, c as pairedDecisionShape, i as PairedMcNemarEvidence, n as PairedDecisionShape, o as PairedPromotionDecisionOptions, r as PairedDecisionStatistic, s as decidePairedPromotion, t as PairedDecisionMethod } from "./paired-promotion-decision-B6zJ3gYM.js";
|
|
26
|
+
import { C as HeldOutGate, E as SplitCoverage, S as GateEvidence, T as HeldOutGateRejectionCode, _ as paretoChart, a as ParetoPoint, b as DeltaStatistic, c as ResearchReportCandidate, d as ResearchReportOptions, f as ResearchReportRecommendation, g as gainHistogram, h as SummaryTableRow, i as ParetoFigureSpec, l as ResearchReportDecision, m as SummaryTableOptions, n as GainDistributionFigureSpec, o as RESEARCH_REPORT_HARD_PAIR_FLOOR, p as SummaryTable, r as GainDistributionOptions, s as ResearchReport, t as GainDistributionBin, u as ResearchReportMethodology, v as researchReport, w as HeldOutGateConfig, x as GateDecision, y as summaryTable } from "./summary-report-DuUS_i7W.js";
|
|
25
27
|
import { a as FailureClassification, c as classifyFailure, i as DEFAULT_RULES, o as FailureContext, s as FailureRule } from "./failure-cluster-CqcvCcdR.js";
|
|
26
|
-
import { C as
|
|
27
|
-
import { A as isTrainableSplit, D as assertRolloutLine, E as assertMintedLines, T as assertMinted, b as RolloutSplit, h as RolloutLine, i as ChatToolCall, j as validateRolloutLine, k as isRolloutLine, n as ChatMessage, o as MintedRolloutLine, p as RolloutCapture, s as MintedRolloutOutcome, u as ROLLOUT_SCHEMA, w as ToolDef, x as RolloutStep, y as RolloutRole } from "./schema-
|
|
28
|
-
import { C as Unavailable, D as showMeasured, E as isUnavailable, S as SupervisorRunTreeGapCode, _ as SupervisorRunReport, b as SupervisorRunTree, f as SUPERVISOR_RUN_SCHEMA, g as SupervisorRunReader, h as SupervisorRunNodeRole, p as SourceLimits, r as Measured, v as SupervisorRunRollup, x as SupervisorRunTreeGap, y as SupervisorRunSources } from "./types-
|
|
29
|
-
import { B as BenchmarkSource, F as BenchmarkDatasetItem, H as deterministicSplit, I as BenchmarkEvaluation, L as BenchmarkFamily, N as BENCHMARK_SPLIT_SEED, P as BenchmarkAdapter, R as BenchmarkResponder, V as BenchmarkTaskKind, t as index_d_exports, z as BenchmarkScenario } from "./index-
|
|
30
|
-
import { $ as
|
|
31
|
-
import { $
|
|
28
|
+
import { C as toOpenAiTool, S as makeEvalTools, _ as TraceAnalysisToolDescriptor, a as TraceAnalystLimits, b as EvalToolDef, d as RawAnalystFinding, g as TRACE_ANALYST_TOOL_NAMESPACE, h as BuildTraceAnalysisToolsOptions, i as TraceAnalysisEngineResult, l as RawAnalystEvidence, n as TraceAnalysisEngine, o as resolveTraceAnalystLimits, r as TraceAnalysisEngineRequest, t as DEFAULT_TRACE_ANALYST_LIMITS, v as buildTraceAnalysisToolDescriptors, x as MakeEvalToolsConfig, y as traceAnalystFunctionGroup } from "./engine-nB64f48I.js";
|
|
29
|
+
import { A as isTrainableSplit, D as assertRolloutLine, E as assertMintedLines, T as assertMinted, b as RolloutSplit, h as RolloutLine, i as ChatToolCall, j as validateRolloutLine, k as isRolloutLine, n as ChatMessage, o as MintedRolloutLine, p as RolloutCapture, s as MintedRolloutOutcome, u as ROLLOUT_SCHEMA, w as ToolDef, x as RolloutStep, y as RolloutRole } from "./schema-Cef2cFmb2.js";
|
|
30
|
+
import { C as Unavailable, D as showMeasured, E as isUnavailable, S as SupervisorRunTreeGapCode, _ as SupervisorRunReport, b as SupervisorRunTree, f as SUPERVISOR_RUN_SCHEMA, g as SupervisorRunReader, h as SupervisorRunNodeRole, p as SourceLimits, r as Measured, v as SupervisorRunRollup, x as SupervisorRunTreeGap, y as SupervisorRunSources } from "./types-D4mog56g.js";
|
|
31
|
+
import { B as BenchmarkSource, F as BenchmarkDatasetItem, H as deterministicSplit, I as BenchmarkEvaluation, L as BenchmarkFamily, N as BENCHMARK_SPLIT_SEED, P as BenchmarkAdapter, R as BenchmarkResponder, V as BenchmarkTaskKind, t as index_d_exports, z as BenchmarkScenario } from "./index-C5HOo4ZF2.js";
|
|
32
|
+
import { $ as redTeamDataset, J as RedTeamCase, Q as RedTeamReport, X as RedTeamFinding, Y as RedTeamCategory, Z as RedTeamPayload, an as ReferenceEquivalenceScenario, at as CanaryKind, cn as LlmJudgeDimension, ct as CanarySeverity, en as REFERENCE_EQUIVALENCE_INPUT_LIMITS, et as redTeamReport, in as ReferenceEquivalenceJudgeResult, it as CanaryEvaluation, ln as LlmJudgeOptions, lt as runCanaries, nn as ReferenceEquivalenceJudgeInput, nt as toolNamesForRun, on as createReferenceEquivalenceJudge, ot as CanaryOptions, q as DEFAULT_RED_TEAM_CORPUS, rn as ReferenceEquivalenceJudgeOptions, rt as CanaryAlert, sn as runReferenceEquivalenceJudge, st as CanaryReport, tn as REFERENCE_EQUIVALENCE_JUDGE_VERSION, tt as scoreRedTeamOutput, un as llmJudge } from "./skillopt-optimization-method-BO7NAl3b.js";
|
|
33
|
+
import { $t as ToolCallEventLike, Gt as BackendIntegrityReport, Jt as summarizeAgentReceiptIntegrity, Kt as assertRealAgentReceipts, Qt as RuntimeEventLike, Wt as BackendIntegrityError, Xt as ArtifactEventLike, Yt as summarizeBackendIntegrity, Zt as ProposalEventLike, en as extractProducedState, qt as assertRealBackend } from "./index-Sh2I0DRc.js";
|
|
34
|
+
import { A as PairArmsResult, C as hashJson, D as MatchedPair, E as ComparePairedArmsOptions, F as PairedMetricDelta, I as comparePairedArms, L as pairArms, M as PairedArmRow, N as PairedArmsComparison, O as MatchedRunRecordPair, P as PairedCorrectness, R as pairRunRecords, S as evaluateHypothesis, T as verifyManifest, _ as HypothesisManifest, b as SignedManifestAlgo, j as PairRunRecordsResult, k as PairArmsOptions, v as HypothesisResult, w as signManifest, x as canonicalize, y as SignedManifest } from "./statistical-heldout-Dn9ruizm.js";
|
|
35
|
+
import { _ as paretoFrontier, f as Direction, g as dominates, h as crowdingDistance, m as ParetoResult, p as Objective, v as paretoFrontierWithCrowding, y as scalarScore } from "./promotion-policy-ChWhTDBH.js";
|
|
32
36
|
import { _ as vitestTestParser, a as DockerSandboxDriver, c as SandboxHarness, d as SubprocessSandboxDriver, f as SubprocessSandboxDriverOptions, g as pytestTestParser, h as jestTestParser, i as runTestGradedScenario, l as SandboxHarnessResult, m as composeParsers, n as TestGradedRunResult, o as HarnessConfig, p as TestOutputParser, r as TestGradedScenario, s as SandboxDriver, t as TestGradedRunOptions, u as SandboxResult } from "./test-graded-scenario-D1TaI2va.js";
|
|
33
|
-
import { $ as ControlRunResult, A as feedbackTrajectoryToOptimizerRow, B as AnalystReviewCounts, C as analystRunToReviewRequests, D as feedbackTrajectoriesToDatasetScenarios, E as createFeedbackTrajectory, F as serializeFeedbackTrajectoriesJsonl, G as analystFindingDigest, H as AnalystReviewQuality, I as summarizePreferenceMemory, J as ControlActionOutcome, K as analystRunDigest, L as withAssignedFeedbackSplit, M as renderPreferenceMemoryMarkdown, N as replayFeedbackTrajectories, O as feedbackTrajectoriesToOptimizerRows, P as replayFeedbackTrajectory, Q as ControlEvalResult, R as AnalystFindingDigest, S as analystRunToFeedbackTrajectory, T as controlRunToFeedbackTrajectory, U as AnalystReviewSource, V as AnalystReviewDecision, W as AnalystRunDigest, X as ControlContext, Y as ControlBudget, Z as ControlDecision, _ as FeedbackTrajectoryStore, a as FeedbackLabel, at as StopDecision, b as PreferenceMemoryEntry, c as FeedbackOptimizerRow, ct as runAgentControlLoop, d as FeedbackReplayResult, dt as subjectiveEval, et as ControlRuntimeConfig, f as FeedbackSeverity, g as FeedbackTrajectoryFilter, h as FeedbackTrajectory, i as FeedbackAttempt, it as ControlStopPolicies, j as parseFeedbackTrajectoriesJsonl, k as feedbackTrajectoryToDatasetScenario, l as FeedbackOutcome, lt as stopOnNoProgress, m as FeedbackTask, n as AnalystReviewRequest, nt as ControlSeverity, o as FeedbackLabelKind, ot as allCriticalPassed, p as FeedbackSplitPolicy, q as ControlActionFailureMode, r as FeedbackArtifactType, rt as ControlStep, s as FeedbackLabelSource, st as objectiveEval, t as AnalystFeedbackTrajectoryOptions, tt as ControlRuntimeError, u as FeedbackReplayAdapter, ut as stopOnRepeatedAction, v as FileSystemFeedbackTrajectoryStore, w as assignFeedbackSplit, x as ProposedSideEffect, y as InMemoryFeedbackTrajectoryStore, z as AnalystMissedIssue } from "./feedback-trajectory-
|
|
34
|
-
import { A as ActionExecutionPolicy, C as ReviewMemoryStore, D as inMemoryReviewStore, E as createLlmReviewer, M as evaluateActionPolicy, O as jsonlReviewStore, S as ReviewMemoryEntry, T as VerifyFn, _ as ProposeReviewReport, a as ProposeReviewControlAction, b as ReviewFn, c as ProposeReviewControlState, d as LlmJsonCall, f as LlmReviewerConfig, g as ProposeReviewConfig, h as ProposeOutput, i as scoreFromEvals, j as ActionPolicyDecision, k as runProposeReview, l as controlFailureClassFromVerification, m as ProposeInput, n as RunEvidenceMetadata, o as ProposeReviewControlConfig, p as ProposeFn, r as controlRunToRunRecord, s as ProposeReviewControlResult, t as ControlRunToRunRecordOptions, u as runProposeReviewAsControlLoop, v as ProposeReviewShot, w as Verification, x as ReviewInput, y as Review } from "./run-evidence-
|
|
35
|
-
import { _ as
|
|
36
|
-
import { $ as ScoreOrigin, $t as toSftRows, Ct as HarborSubagentTrajectoryRef, Dt as relabelImportedSplit, Et as fromHarborTrajectory, Gt as SftRow, Ht as RewardRow, J as MintRolloutOptions, Ot as toHarborTrajectories, Q as unmintableReasons, St as HarborStepSource, Tt as HarborTrajectory, Wt as SftExportOptions, X as RolloutScrubber, Xt as toRewardRows, Y as MintRolloutResult, Yt as toJsonl, Z as mintRolloutRows, _t as HarborImageSource, at as trainingReward, bt as HarborObservationResult, dt as ATIF_SCHEMA_VERSION, et as ScorePreference, ft as FromHarborOptions, gt as HarborFinalMetrics, ht as HarborContentPart, it as scoreOrigin, kt as toHarborTrajectory, mt as HarborAgent, nt as observedScore, ot as trainingScore, pt as HARBOR_IMPORT_GAP, rt as observedSplitScore, tt as isRealnessGated, vt as HarborMetrics, wt as HarborToolCall, xt as HarborStep, yt as HarborObservation } from "./index-B6-B0zTB.js";
|
|
37
|
-
import { C as SupervisorRunIntegrityIssue, D as SupervisorRunIntegritySeverity, E as SupervisorRunIntegrityReport, F as analyzeSupervisorRunSources, L as rollupSupervisorRuns, N as claudeCodeSupervisorRunReader, O as SupervisorRunSourceOnlyCheckCode, P as readClaudeCodeSupervisorRun, S as SupervisorRunIntegrityEvidence, T as SupervisorRunIntegrityOptions, a as supervisorRunRolloutLines, b as analyzeSupervisorRunIntegrity, c as renderSupervisorRunMarkdown, d as analyzeSupervisorRun, n as readRuntimeSupervisorRun, r as runtimeSupervisorRunReader, s as renderSupervisorRunHeadline, t as isRuntimeSupervisorRunDir, v as writeSupervisorRunReport, w as SupervisorRunIntegrityIssueCode, x as SUPERVISOR_RUN_INTEGRITY_SCHEMA } from "./index-BZUe-ODI.js";
|
|
38
|
-
import { a as WelchTestResult, c as iqr, d as TrajectoryStep, f as buildTrajectory, g as computeToolUseMetrics, h as ToolUseOptions, i as MetricVerdict, l as welchsTTest, m as ToolUseMetrics, n as BaselineReport, o as WelchTestStatus, p as ToolStats, r as MetricSamples, s as compareToBaseline, t as BaselineOptions, u as Trajectory } from "./baseline-D_fT6277.js";
|
|
39
|
-
import { n as SeriesConvergenceResult, r as analyzeSeries, t as SeriesConvergenceOptions } from "./series-convergence-ofsqPWhs.js";
|
|
40
|
-
import { _ as EvalCampaignResult, a as FailureMode, c as SteeringChange, d as CampaignRunContext, f as CampaignRunOutcome, g as EvalCampaignOptions, h as CampaignVariant, i as ExperimentResult, l as CampaignFactoryParams, m as CampaignScenario, n as CallbackResearcherOptions, o as NoopResearcher, p as CampaignRunner, r as ExperimentPlan, s as Researcher, t as CallbackResearcher, u as CampaignIntegrityPolicy, v as FailedRun, y as runEvalCampaign } from "./researcher-xLeNcpKX.js";
|
|
37
|
+
import { $ as ControlRunResult, A as feedbackTrajectoryToOptimizerRow, B as AnalystReviewCounts, C as analystRunToReviewRequests, D as feedbackTrajectoriesToDatasetScenarios, E as createFeedbackTrajectory, F as serializeFeedbackTrajectoriesJsonl, G as analystFindingDigest, H as AnalystReviewQuality, I as summarizePreferenceMemory, J as ControlActionOutcome, K as analystRunDigest, L as withAssignedFeedbackSplit, M as renderPreferenceMemoryMarkdown, N as replayFeedbackTrajectories, O as feedbackTrajectoriesToOptimizerRows, P as replayFeedbackTrajectory, Q as ControlEvalResult, R as AnalystFindingDigest, S as analystRunToFeedbackTrajectory, T as controlRunToFeedbackTrajectory, U as AnalystReviewSource, V as AnalystReviewDecision, W as AnalystRunDigest, X as ControlContext, Y as ControlBudget, Z as ControlDecision, _ as FeedbackTrajectoryStore, a as FeedbackLabel, at as StopDecision, b as PreferenceMemoryEntry, c as FeedbackOptimizerRow, ct as runAgentControlLoop, d as FeedbackReplayResult, dt as subjectiveEval, et as ControlRuntimeConfig, f as FeedbackSeverity, g as FeedbackTrajectoryFilter, h as FeedbackTrajectory, i as FeedbackAttempt, it as ControlStopPolicies, j as parseFeedbackTrajectoriesJsonl, k as feedbackTrajectoryToDatasetScenario, l as FeedbackOutcome, lt as stopOnNoProgress, m as FeedbackTask, n as AnalystReviewRequest, nt as ControlSeverity, o as FeedbackLabelKind, ot as allCriticalPassed, p as FeedbackSplitPolicy, q as ControlActionFailureMode, r as FeedbackArtifactType, rt as ControlStep, s as FeedbackLabelSource, st as objectiveEval, t as AnalystFeedbackTrajectoryOptions, tt as ControlRuntimeError, u as FeedbackReplayAdapter, ut as stopOnRepeatedAction, v as FileSystemFeedbackTrajectoryStore, w as assignFeedbackSplit, x as ProposedSideEffect, y as InMemoryFeedbackTrajectoryStore, z as AnalystMissedIssue } from "./feedback-trajectory-Rh280oXo.js";
|
|
38
|
+
import { A as ActionExecutionPolicy, C as ReviewMemoryStore, D as inMemoryReviewStore, E as createLlmReviewer, M as evaluateActionPolicy, O as jsonlReviewStore, S as ReviewMemoryEntry, T as VerifyFn, _ as ProposeReviewReport, a as ProposeReviewControlAction, b as ReviewFn, c as ProposeReviewControlState, d as LlmJsonCall, f as LlmReviewerConfig, g as ProposeReviewConfig, h as ProposeOutput, i as scoreFromEvals, j as ActionPolicyDecision, k as runProposeReview, l as controlFailureClassFromVerification, m as ProposeInput, n as RunEvidenceMetadata, o as ProposeReviewControlConfig, p as ProposeFn, r as controlRunToRunRecord, s as ProposeReviewControlResult, t as ControlRunToRunRecordOptions, u as runProposeReviewAsControlLoop, v as ProposeReviewShot, w as Verification, x as ReviewInput, y as Review } from "./run-evidence-BDFFai9R.js";
|
|
39
|
+
import { C as ClusteredPairedBinaryOptions, E as clusteredPairedBinary, S as ClusteredMatchedPair, T as ClusteredPairedBinaryStatistics, _ as inMemoryExperimentStore, a as ExperimentStats, b as ClusterSignFlipResult, c as ExperimentTrackerOptions, d as ImprovementVerdictResult, f as ProvenanceReader, g as improvementVerdict, h as gitProvenanceReader, i as ExperimentRep, l as ExperimentVerdict, m as fileExperimentStore, n as Experiment, o as ExperimentStore, p as computeExperimentStats, r as ExperimentProvenance, s as ExperimentTracker, t as CreateExperimentInput, u as ImprovementThresholds, v as ClusterBootstrapInterval, w as ClusteredPairedBinaryResult, x as ClusteredBinaryCluster, y as ClusterSignFlipAlternative } from "./experiment-tracker-IMntXr6J.js";
|
|
41
40
|
import { a as PairedEvalueStep, c as pairedEvalueSequence, i as PairedEvalueSequence, n as InterimReleaseConfidenceInput, o as SequentialDecision, r as PairedEvalueOptions, s as evaluateInterimReleaseConfidence, t as InterimReleaseConfidence } from "./sequential-CYwq6Ff_.js";
|
|
41
|
+
import { _ as ReleaseConfidenceStatus, a as JudgeReplayGateArgs, b as assertReleaseConfidence, c as judgeReplayGate, d as ReleaseConfidenceAxis, f as ReleaseConfidenceAxisName, g as ReleaseConfidenceScorecard, h as ReleaseConfidenceMetrics, i as BootstrapResult, l as ActionableSideInfo, m as ReleaseConfidenceIssue, n as renderReleaseReport, o as Verdict, p as ReleaseConfidenceInput, r as BootstrapOptions, s as bootstrapCi, t as RenderReleaseReportOptions, u as AsiSeverity, v as ReleaseConfidenceThresholds, x as evaluateReleaseConfidence, y as ReleaseTraceEvidence } from "./release-report-CI8uisI1.js";
|
|
42
|
+
import { $ as ScoreOrigin, $t as toSftRows, Ct as HarborSubagentTrajectoryRef, Dt as relabelImportedSplit, Et as fromHarborTrajectory, Gt as SftRow, Ht as RewardRow, J as MintRolloutOptions, Ot as toHarborTrajectories, Q as unmintableReasons, St as HarborStepSource, Tt as HarborTrajectory, Wt as SftExportOptions, X as RolloutScrubber, Xt as toRewardRows, Y as MintRolloutResult, Yt as toJsonl, Z as mintRolloutRows, _t as HarborImageSource, at as trainingReward, bt as HarborObservationResult, dt as ATIF_SCHEMA_VERSION, et as ScorePreference, ft as FromHarborOptions, gt as HarborFinalMetrics, ht as HarborContentPart, it as scoreOrigin, kt as toHarborTrajectory, mt as HarborAgent, nt as observedScore, ot as trainingScore, pt as HARBOR_IMPORT_GAP, rt as observedSplitScore, tt as isRealnessGated, vt as HarborMetrics, wt as HarborToolCall, xt as HarborStep, yt as HarborObservation } from "./index-CvXXlyz7.js";
|
|
43
|
+
import { C as SupervisorRunIntegrityIssue, D as SupervisorRunIntegritySeverity, E as SupervisorRunIntegrityReport, F as analyzeSupervisorRunSources, L as rollupSupervisorRuns, N as claudeCodeSupervisorRunReader, O as SupervisorRunSourceOnlyCheckCode, P as readClaudeCodeSupervisorRun, S as SupervisorRunIntegrityEvidence, T as SupervisorRunIntegrityOptions, a as supervisorRunRolloutLines, b as analyzeSupervisorRunIntegrity, c as renderSupervisorRunMarkdown, d as analyzeSupervisorRun, n as readRuntimeSupervisorRun, r as runtimeSupervisorRunReader, s as renderSupervisorRunHeadline, t as isRuntimeSupervisorRunDir, v as writeSupervisorRunReport, w as SupervisorRunIntegrityIssueCode, x as SUPERVISOR_RUN_INTEGRITY_SCHEMA } from "./index-BZ3-y4YL.js";
|
|
44
|
+
import { a as WelchTestResult, c as iqr, d as ToolUseMetrics, f as ToolUseOptions, i as MetricVerdict, l as welchsTTest, n as BaselineReport, o as WelchTestStatus, p as computeToolUseMetrics, r as MetricSamples, s as compareToBaseline, t as BaselineOptions, u as ToolStats } from "./baseline-CavEbRyH.js";
|
|
45
|
+
import { n as TrajectoryStep, r as buildTrajectory, t as Trajectory } from "./trajectory-YC15QDYQ.js";
|
|
46
|
+
import { n as SeriesConvergenceResult, r as analyzeSeries, t as SeriesConvergenceOptions } from "./series-convergence-ofsqPWhs.js";
|
|
47
|
+
import { a as attributeCounterfactuals, i as CounterfactualRunner, n as CounterfactualMutation, o as runCounterfactual, r as CounterfactualResult, t as CounterfactualContext } from "./counterfactual-CxmxAONP.js";
|
|
48
|
+
import { _ as EvalCampaignResult, a as FailureMode, c as SteeringChange, d as CampaignRunContext, f as CampaignRunOutcome, g as EvalCampaignOptions, h as CampaignVariant, i as ExperimentResult, l as CampaignFactoryParams, m as CampaignScenario, n as CallbackResearcherOptions, o as NoopResearcher, p as CampaignRunner, r as ExperimentPlan, s as Researcher, t as CallbackResearcher, u as CampaignIntegrityPolicy, v as FailedRun, y as runEvalCampaign } from "./researcher-BoaxeCzP.js";
|
|
42
49
|
//#region src/auto-pr.d.ts
|
|
43
50
|
/**
|
|
44
51
|
* Automated pull-request transports for the production loop.
|
|
@@ -354,124 +361,6 @@ declare class ProductClient {
|
|
|
354
361
|
*/
|
|
355
362
|
declare function runE2EWorkflow(client: ProductClient, name: string, workflow: (client: ProductClient) => Promise<CheckResult[]>): Promise<TestResult>;
|
|
356
363
|
//#endregion
|
|
357
|
-
//#region src/clustered-paired-binary.d.ts
|
|
358
|
-
/**
|
|
359
|
-
* Paired binary comparison for work items nested inside independent clusters.
|
|
360
|
-
*
|
|
361
|
-
* Pairing is delegated to {@link pairArms}; this module adds the cluster-aware
|
|
362
|
-
* estimands and inference that task-level McNemar/bootstrap utilities cannot
|
|
363
|
-
* provide. Callers keep their own row shape through accessors, and every
|
|
364
|
-
* matched or unpaired result returns the original row object unchanged.
|
|
365
|
-
*/
|
|
366
|
-
type ClusterSignFlipAlternative = 'two-sided' | 'greater' | 'less';
|
|
367
|
-
interface ClusteredPairedBinaryOptions<TRow> {
|
|
368
|
-
/** Arm treated as the control side of every pair. */
|
|
369
|
-
baselineArm: string;
|
|
370
|
-
/** Arm treated as the treatment side of every pair. */
|
|
371
|
-
treatmentArm: string;
|
|
372
|
-
/** Stable work-item identity shared by both arms. */
|
|
373
|
-
pairKey: (row: TRow) => string;
|
|
374
|
-
/** Independent-cluster identity, shared by both rows in a matched pair. */
|
|
375
|
-
clusterKey: (row: TRow) => string;
|
|
376
|
-
/** Arm identity for this row. Rows from other arms are ignored. */
|
|
377
|
-
arm: (row: TRow) => string;
|
|
378
|
-
/** Binary outcome. */
|
|
379
|
-
pass: (row: TRow) => boolean;
|
|
380
|
-
/** Replicate identity when a work item has multiple rows in either arm. */
|
|
381
|
-
repKey?: (row: TRow) => string | undefined;
|
|
382
|
-
/** Deterministic seed for bootstrap and Monte Carlo sign-flip draws. */
|
|
383
|
-
seed: number;
|
|
384
|
-
/** Percentile confidence level. Default 0.95. */
|
|
385
|
-
confidence?: number;
|
|
386
|
-
/** Whole-cluster bootstrap draws. Default 10,000; maximum 1,000,000. */
|
|
387
|
-
bootstrapResamples?: number;
|
|
388
|
-
/** Sign-flip alternative. Default 'two-sided'. */
|
|
389
|
-
alternative?: ClusterSignFlipAlternative;
|
|
390
|
-
/**
|
|
391
|
-
* Enumerate every sign assignment at or below this many non-zero clusters;
|
|
392
|
-
* otherwise use Monte Carlo. Default 20; maximum 20.
|
|
393
|
-
*/
|
|
394
|
-
exactClusterLimit?: number;
|
|
395
|
-
/** Monte Carlo sign-flip draws when exact enumeration is not used. Default 100,000; maximum 1,000,000. */
|
|
396
|
-
signFlipResamples?: number;
|
|
397
|
-
}
|
|
398
|
-
interface ClusteredMatchedPair<TRow> {
|
|
399
|
-
pairKey: string;
|
|
400
|
-
repIndex: number;
|
|
401
|
-
clusterKey: string;
|
|
402
|
-
baseline: TRow;
|
|
403
|
-
treatment: TRow;
|
|
404
|
-
baselinePass: boolean;
|
|
405
|
-
treatmentPass: boolean;
|
|
406
|
-
}
|
|
407
|
-
interface ClusteredBinaryCluster {
|
|
408
|
-
clusterKey: string;
|
|
409
|
-
nPairs: number;
|
|
410
|
-
/** Treatment passes and baseline fails. */
|
|
411
|
-
b10: number;
|
|
412
|
-
/** Baseline passes and treatment fails. */
|
|
413
|
-
b01: number;
|
|
414
|
-
/** Mean (treatment - baseline) binary outcome within this cluster. */
|
|
415
|
-
meanDifference: number;
|
|
416
|
-
}
|
|
417
|
-
interface ClusterBootstrapInterval {
|
|
418
|
-
/** The interval resamples clusters and recomputes this task-weighted statistic. */
|
|
419
|
-
statistic: 'task-weighted-risk-difference';
|
|
420
|
-
lower: number;
|
|
421
|
-
upper: number;
|
|
422
|
-
confidence: number;
|
|
423
|
-
resamples: number;
|
|
424
|
-
seed: number;
|
|
425
|
-
}
|
|
426
|
-
interface ClusterSignFlipResult {
|
|
427
|
-
/** Task-weighted paired risk difference, matching the reported bootstrap estimand. */
|
|
428
|
-
statistic: number;
|
|
429
|
-
/**
|
|
430
|
-
* Randomization p-value under whole-cluster arm-label exchangeability.
|
|
431
|
-
* `method: 'exact'` means every cluster-level sign assignment was enumerated;
|
|
432
|
-
* it does not make the exchangeability assumption unnecessary.
|
|
433
|
-
*/
|
|
434
|
-
pValue: number;
|
|
435
|
-
alternative: ClusterSignFlipAlternative;
|
|
436
|
-
method: 'exact' | 'monte-carlo';
|
|
437
|
-
/** Exact assignments enumerated or Monte Carlo assignments drawn. */
|
|
438
|
-
assignments: number;
|
|
439
|
-
nClusters: number;
|
|
440
|
-
nNonZeroClusters: number;
|
|
441
|
-
/** Null for exact enumeration, which has no random draws. */
|
|
442
|
-
seed: number | null;
|
|
443
|
-
}
|
|
444
|
-
interface ClusteredPairedBinaryStatistics {
|
|
445
|
-
nPairs: number;
|
|
446
|
-
nClusters: number;
|
|
447
|
-
/** Treatment passes and baseline fails, across all matched pairs. */
|
|
448
|
-
b10: number;
|
|
449
|
-
/** Baseline passes and treatment fails, across all matched pairs. */
|
|
450
|
-
b01: number;
|
|
451
|
-
/** Mean paired difference across tasks, so every task has equal weight. */
|
|
452
|
-
taskWeightedRiskDifference: number;
|
|
453
|
-
/** Mean of cluster-level paired differences, so every cluster has equal weight. */
|
|
454
|
-
equalClusterMean: number;
|
|
455
|
-
clusters: ClusteredBinaryCluster[];
|
|
456
|
-
/** Null below two independent clusters; one cluster cannot estimate cluster uncertainty. */
|
|
457
|
-
bootstrap: ClusterBootstrapInterval | null;
|
|
458
|
-
signFlip: ClusterSignFlipResult;
|
|
459
|
-
}
|
|
460
|
-
interface ClusteredPairedBinaryResult<TRow> {
|
|
461
|
-
matchedPairs: ClusteredMatchedPair<TRow>[];
|
|
462
|
-
unpairedBaseline: TRow[];
|
|
463
|
-
unpairedTreatment: TRow[];
|
|
464
|
-
/** Null when there are no matched rows; absence is never reported as a zero effect. */
|
|
465
|
-
statistics: ClusteredPairedBinaryStatistics | null;
|
|
466
|
-
}
|
|
467
|
-
/**
|
|
468
|
-
* Compare binary outcomes on matched work items while respecting independent
|
|
469
|
-
* clusters. The confidence interval resamples whole clusters and recomputes the
|
|
470
|
-
* task-weighted risk difference. The sign-flip test flips whole-cluster outcome
|
|
471
|
-
* totals and tests that same task-weighted estimand.
|
|
472
|
-
*/
|
|
473
|
-
declare function clusteredPairedBinary<TRow>(rows: readonly TRow[], options: ClusteredPairedBinaryOptions<TRow>): ClusteredPairedBinaryResult<TRow>;
|
|
474
|
-
//#endregion
|
|
475
364
|
//#region src/deprecation.d.ts
|
|
476
365
|
/**
|
|
477
366
|
* One-shot deprecation warnings. A deprecated surface warns the FIRST time it
|
|
@@ -678,12 +567,22 @@ interface ModelPreflight {
|
|
|
678
567
|
model: string;
|
|
679
568
|
/** Membership in the `{baseUrl}/models` served set. */
|
|
680
569
|
listed: boolean;
|
|
681
|
-
/**
|
|
570
|
+
/**
|
|
571
|
+
* The router reached a provider for this model. `null` when `probe` was not
|
|
572
|
+
* requested. True on a 2xx, and also when the provider answered by
|
|
573
|
+
* exhausting its reasoning budget — that proves reachability while leaving
|
|
574
|
+
* identity unproven (see `budgetExhausted`).
|
|
575
|
+
*/
|
|
682
576
|
served: boolean | null;
|
|
683
577
|
/** HTTP status of the probe. `null` when not probed. */
|
|
684
578
|
status: number | null;
|
|
685
579
|
/** Probe body's `error.message` when present, else `null`. */
|
|
686
580
|
detail: string | null;
|
|
581
|
+
/**
|
|
582
|
+
* The probe died on the token budget rather than on the model. Identity is
|
|
583
|
+
* unproven, not refuted: raise `probeMaxTokens` to resolve it.
|
|
584
|
+
*/
|
|
585
|
+
budgetExhausted: boolean;
|
|
687
586
|
/**
|
|
688
587
|
* Identity of the model that answered the probe, compared against `model`.
|
|
689
588
|
* `null` when the model was not probed or the probe failed.
|
|
@@ -697,8 +596,14 @@ interface PreflightModelsOptions {
|
|
|
697
596
|
apiKey: string;
|
|
698
597
|
/** Model ids to check. */
|
|
699
598
|
models: string[];
|
|
700
|
-
/** When true, additionally spend a
|
|
599
|
+
/** When true, additionally spend a small chat probe per model. Default false. */
|
|
701
600
|
probe?: boolean;
|
|
601
|
+
/**
|
|
602
|
+
* Output-token budget per probe. Default `PROBE_MAX_TOKENS`. Lowering it
|
|
603
|
+
* below the reasoning floor makes healthy reasoning models report
|
|
604
|
+
* `budgetExhausted` instead of proving their identity.
|
|
605
|
+
*/
|
|
606
|
+
probeMaxTokens?: number;
|
|
702
607
|
/** Injectable fetch for tests; defaults to the global. */
|
|
703
608
|
fetchImpl?: typeof fetch;
|
|
704
609
|
}
|
|
@@ -714,7 +619,7 @@ interface PreflightOutcome {
|
|
|
714
619
|
* fallbacks.
|
|
715
620
|
*
|
|
716
621
|
* The membership check (one GET) always runs. When `probe` is true, each model
|
|
717
|
-
* additionally gets a
|
|
622
|
+
* additionally gets a small chat probe so a model that is listed but
|
|
718
623
|
* unconfigured (a 401 `model_not_found` from the router) is caught.
|
|
719
624
|
*/
|
|
720
625
|
declare function preflightModels(opts: PreflightModelsOptions): Promise<PreflightOutcome>;
|
|
@@ -2436,185 +2341,6 @@ declare class EvalTraceStore {
|
|
|
2436
2341
|
compareRuns(candidateA: string, candidateB: string): Promise<CandidateComparison>;
|
|
2437
2342
|
}
|
|
2438
2343
|
//#endregion
|
|
2439
|
-
//#region src/experiment-tracker.d.ts
|
|
2440
|
-
/**
|
|
2441
|
-
* Experiment tracker — git-provenanced experiment log with N-rep stats and a
|
|
2442
|
-
* KEEP / REGRESSION / NOISE verdict against a parent.
|
|
2443
|
-
*
|
|
2444
|
-
* Every loop the fleet runs reduces to the same question: "I ran the candidate
|
|
2445
|
-
* N times — is the median measurably better than the parent, or is the delta
|
|
2446
|
-
* inside the noise band?" The hand-rolled copies bake a fixed score scale
|
|
2447
|
-
* (percentage points), a fixed store path (`.evolve/experiments-v2.json`), and
|
|
2448
|
-
* `execSync('git …')` straight into the module. This is the canonical version:
|
|
2449
|
-
* provenance and persistence are injected, thresholds are configurable, and the
|
|
2450
|
-
* stats + verdict are pure functions you can unit-test without a git repo or a
|
|
2451
|
-
* filesystem.
|
|
2452
|
-
*
|
|
2453
|
-
* Stats per experiment: median / mean / min / max / iqr / stddev / passRate /
|
|
2454
|
-
* n, plus a `stable` flag (`iqr < iqrUnstableAbove && stddev < stddevUnstableAbove`).
|
|
2455
|
-
*
|
|
2456
|
-
* Verdict against a parent (both must have `n >= minRepsForVerdict`):
|
|
2457
|
-
* - NOISE — the candidate is too unstable to judge (`!stable`)
|
|
2458
|
-
* - KEEP — `medianDelta > keepThreshold`
|
|
2459
|
-
* - REGRESSION — `medianDelta < -regressionThreshold`
|
|
2460
|
-
* - NOISE — otherwise (delta inside the band)
|
|
2461
|
-
* With no parent (or insufficient reps) the verdict is the neutral ITERATE.
|
|
2462
|
-
*/
|
|
2463
|
-
/** Verdict for one experiment relative to its parent. ITERATE is the neutral
|
|
2464
|
-
* "keep collecting reps / no parent to compare against" state. */
|
|
2465
|
-
type ExperimentVerdict = 'KEEP' | 'ITERATE' | 'NOISE' | 'REGRESSION';
|
|
2466
|
-
/** Git provenance for the working tree an experiment was run from. */
|
|
2467
|
-
interface ExperimentProvenance {
|
|
2468
|
-
/** Commit sha (short or full — the tracker does not interpret it). */
|
|
2469
|
-
commit: string;
|
|
2470
|
-
/** First line of the commit message. */
|
|
2471
|
-
message: string;
|
|
2472
|
-
/** Files changed vs the parent commit, or a marker like 'uncommitted'. */
|
|
2473
|
-
changedFiles: string[];
|
|
2474
|
-
}
|
|
2475
|
-
/** A single repetition of an experiment, carrying the score the verdict is
|
|
2476
|
-
* computed on plus any free-form per-rep metrics the consumer wants kept. */
|
|
2477
|
-
interface ExperimentRep {
|
|
2478
|
-
/** 0-indexed repetition number within the experiment. */
|
|
2479
|
-
rep: number;
|
|
2480
|
-
/** The score this rep is judged on (same scale as the thresholds). */
|
|
2481
|
-
score: number;
|
|
2482
|
-
/** ISO timestamp the rep completed. */
|
|
2483
|
-
timestamp: string;
|
|
2484
|
-
/** Whether this rep passed the consumer's own gate — folded into `passRate`. */
|
|
2485
|
-
passed?: boolean;
|
|
2486
|
-
/** Free-form numeric metrics retained for later analysis. */
|
|
2487
|
-
metrics?: Record<string, number>;
|
|
2488
|
-
}
|
|
2489
|
-
interface ExperimentStats {
|
|
2490
|
-
median: number;
|
|
2491
|
-
mean: number;
|
|
2492
|
-
min: number;
|
|
2493
|
-
max: number;
|
|
2494
|
-
/** Inter-quartile range of the rep scores. */
|
|
2495
|
-
iqr: number;
|
|
2496
|
-
/** Population standard deviation of the rep scores. */
|
|
2497
|
-
stddev: number;
|
|
2498
|
-
/** Fraction of reps with `passed === true`, over reps that set `passed`.
|
|
2499
|
-
* null when no rep declared a pass/fail outcome. */
|
|
2500
|
-
passRate: number | null;
|
|
2501
|
-
/** Number of reps. */
|
|
2502
|
-
n: number;
|
|
2503
|
-
/** True when the sample is tight enough to trust for a verdict. */
|
|
2504
|
-
stable: boolean;
|
|
2505
|
-
}
|
|
2506
|
-
interface Experiment {
|
|
2507
|
-
/** Stable id for the experiment. */
|
|
2508
|
-
id: string;
|
|
2509
|
-
/** Free-form label / config descriptor. */
|
|
2510
|
-
label: string;
|
|
2511
|
-
/** Git provenance captured when the experiment was created. */
|
|
2512
|
-
provenance: ExperimentProvenance;
|
|
2513
|
-
/** Parent experiment id this candidate is compared against, if any. */
|
|
2514
|
-
parentId?: string;
|
|
2515
|
-
/** One-line summary of what changed from the parent. */
|
|
2516
|
-
changeSummary: string;
|
|
2517
|
-
reps: ExperimentRep[];
|
|
2518
|
-
stats: ExperimentStats;
|
|
2519
|
-
verdict: ExperimentVerdict;
|
|
2520
|
-
/** ISO timestamp the experiment was created. */
|
|
2521
|
-
createdAt: string;
|
|
2522
|
-
}
|
|
2523
|
-
interface ImprovementThresholds {
|
|
2524
|
-
/** medianDelta strictly above this ⇒ KEEP. Default 5. */
|
|
2525
|
-
keepThreshold?: number;
|
|
2526
|
-
/** medianDelta strictly below the negative of this ⇒ REGRESSION. Default 5. */
|
|
2527
|
-
regressionThreshold?: number;
|
|
2528
|
-
/** iqr at or above this ⇒ unstable. Default 10. */
|
|
2529
|
-
iqrUnstableAbove?: number;
|
|
2530
|
-
/** stddev at or above this ⇒ unstable. Default Infinity (iqr-only stability). */
|
|
2531
|
-
stddevUnstableAbove?: number;
|
|
2532
|
-
/** Reps required on BOTH candidate and parent before a verdict is rendered.
|
|
2533
|
-
* Default 3. */
|
|
2534
|
-
minRepsForVerdict?: number;
|
|
2535
|
-
}
|
|
2536
|
-
interface ImprovementVerdictResult {
|
|
2537
|
-
verdict: ExperimentVerdict;
|
|
2538
|
-
/** candidate.median − parent.median; null when no parent or insufficient reps. */
|
|
2539
|
-
medianDelta: number | null;
|
|
2540
|
-
/** Human-readable reason for the verdict — for dashboards and logs. */
|
|
2541
|
-
reason: string;
|
|
2542
|
-
}
|
|
2543
|
-
/**
|
|
2544
|
-
* Compute the N-rep statistics for a set of reps. Pure — no I/O. The `stable`
|
|
2545
|
-
* flag is the trust gate the verdict depends on: a sample whose spread exceeds
|
|
2546
|
-
* the configured bounds can't distinguish a real delta from run-to-run noise.
|
|
2547
|
-
*/
|
|
2548
|
-
declare function computeExperimentStats(reps: ExperimentRep[], thresholds?: ImprovementThresholds): ExperimentStats;
|
|
2549
|
-
/**
|
|
2550
|
-
* Verdict for a candidate against its parent. Pure — operates on already-computed
|
|
2551
|
-
* stats. KEEP/REGRESSION require both sides to have `>= minRepsForVerdict` reps
|
|
2552
|
-
* AND the candidate to be `stable`; otherwise the result is NOISE (unstable) or
|
|
2553
|
-
* ITERATE (not enough reps / no parent).
|
|
2554
|
-
*/
|
|
2555
|
-
declare function improvementVerdict(candidate: ExperimentStats, parent: ExperimentStats | null, thresholds?: ImprovementThresholds): ImprovementVerdictResult;
|
|
2556
|
-
/** Reads git provenance for the working tree. Inject a fake in tests; the
|
|
2557
|
-
* default implementation shells out to `git`. */
|
|
2558
|
-
type ProvenanceReader = () => ExperimentProvenance | Promise<ExperimentProvenance>;
|
|
2559
|
-
/** Persistence seam for the experiment log. Inject in-memory in tests; the
|
|
2560
|
-
* filesystem implementation is `fileExperimentStore`. */
|
|
2561
|
-
interface ExperimentStore {
|
|
2562
|
-
load(): Promise<Experiment[]>;
|
|
2563
|
-
save(experiments: Experiment[]): Promise<void>;
|
|
2564
|
-
}
|
|
2565
|
-
/**
|
|
2566
|
-
* Default provenance reader: `git rev-parse HEAD`, the subject line, and the
|
|
2567
|
-
* files changed vs `HEAD~1`. Fail-loud — a tracker that silently logs
|
|
2568
|
-
* `commit: 'unknown'` corrupts the provenance the whole point of the log is to
|
|
2569
|
-
* carry. When the working tree genuinely has no parent commit, pass an override.
|
|
2570
|
-
*/
|
|
2571
|
-
declare const gitProvenanceReader: ProvenanceReader;
|
|
2572
|
-
/** In-memory store — the default when no persistence is wanted (tests, ephemeral
|
|
2573
|
-
* runs). State lives on the instance. */
|
|
2574
|
-
declare function inMemoryExperimentStore(initial?: Experiment[]): ExperimentStore;
|
|
2575
|
-
/** Filesystem store — a single JSON array at `path`, created on first save. */
|
|
2576
|
-
declare function fileExperimentStore(path: string): ExperimentStore;
|
|
2577
|
-
interface ExperimentTrackerOptions {
|
|
2578
|
-
store?: ExperimentStore;
|
|
2579
|
-
provenanceReader?: ProvenanceReader;
|
|
2580
|
-
thresholds?: ImprovementThresholds;
|
|
2581
|
-
/** Clock seam for deterministic timestamps in tests. Default `Date.now`. */
|
|
2582
|
-
now?: () => number;
|
|
2583
|
-
}
|
|
2584
|
-
interface CreateExperimentInput {
|
|
2585
|
-
id: string;
|
|
2586
|
-
label: string;
|
|
2587
|
-
changeSummary: string;
|
|
2588
|
-
parentId?: string;
|
|
2589
|
-
/** Override provenance instead of reading from git (e.g. CI metadata). */
|
|
2590
|
-
provenance?: ExperimentProvenance;
|
|
2591
|
-
}
|
|
2592
|
-
/**
|
|
2593
|
-
* Stateful tracker over an `ExperimentStore`. Create an experiment (provenance
|
|
2594
|
-
* is captured once), append reps as they complete (stats + verdict recompute on
|
|
2595
|
-
* every append), and read the log back for a dashboard. All persistence and git
|
|
2596
|
-
* access flow through the injected seams, so the tracker is fully testable
|
|
2597
|
-
* without a repo or disk.
|
|
2598
|
-
*/
|
|
2599
|
-
declare class ExperimentTracker {
|
|
2600
|
-
private readonly store;
|
|
2601
|
-
private readonly provenanceReader;
|
|
2602
|
-
private readonly thresholds;
|
|
2603
|
-
private readonly now;
|
|
2604
|
-
constructor(options?: ExperimentTrackerOptions);
|
|
2605
|
-
create(input: CreateExperimentInput): Promise<Experiment>;
|
|
2606
|
-
/** Append a rep (its `rep` index defaults to the current rep count) and
|
|
2607
|
-
* recompute stats + verdict. Returns the updated experiment. */
|
|
2608
|
-
addRep(experimentId: string, rep: Omit<ExperimentRep, 'rep' | 'timestamp'> & {
|
|
2609
|
-
rep?: number;
|
|
2610
|
-
timestamp?: string;
|
|
2611
|
-
}): Promise<Experiment>;
|
|
2612
|
-
get(experimentId: string): Promise<Experiment | undefined>;
|
|
2613
|
-
list(): Promise<Experiment[]>;
|
|
2614
|
-
/** Full verdict (not just the enum) for an experiment vs its parent. */
|
|
2615
|
-
verdictFor(experimentId: string): Promise<ImprovementVerdictResult>;
|
|
2616
|
-
}
|
|
2617
|
-
//#endregion
|
|
2618
2344
|
//#region src/leaderboard.d.ts
|
|
2619
2345
|
interface LeaderboardRow {
|
|
2620
2346
|
/** Group key — the agent-profile cellId by default, else a caller dimension. */
|
|
@@ -3092,6 +2818,151 @@ declare function collectionPreserved<T, K extends keyof T & string>(key: K, minR
|
|
|
3092
2818
|
/** Common check: a status field advanced in an expected order. */
|
|
3093
2819
|
declare function statusAdvanced<T extends Record<string, unknown>>(key: keyof T & string, progression: readonly string[]): ContinuityCheck<T>;
|
|
3094
2820
|
//#endregion
|
|
2821
|
+
//#region src/equivalence-check.d.ts
|
|
2822
|
+
/**
|
|
2823
|
+
* The two-arm blind design. `arms: 2` and `blind: true` are literal types
|
|
2824
|
+
* on purpose: a wider design is a different protocol, not a parameter.
|
|
2825
|
+
*/
|
|
2826
|
+
interface EquivalenceCheckSpec {
|
|
2827
|
+
/**
|
|
2828
|
+
* The strategy member whose obligation the check discharges — the pilot
|
|
2829
|
+
* used `'proof-kernel'`: the Lean kernel proved the two statements
|
|
2830
|
+
* equivalent. The bound checker must declare the same member.
|
|
2831
|
+
*/
|
|
2832
|
+
source: VerificationStrategySource;
|
|
2833
|
+
/**
|
|
2834
|
+
* Stable identity of the claim under verification: a path, digest, or
|
|
2835
|
+
* citation. Pilot: arXiv:1507.05650 inequality (4.6) plus the
|
|
2836
|
+
* five-atom counterexample record.
|
|
2837
|
+
*/
|
|
2838
|
+
artifact: string;
|
|
2839
|
+
/** Exactly two independent arms. An N-arm design is not this protocol. */
|
|
2840
|
+
arms: 2;
|
|
2841
|
+
/**
|
|
2842
|
+
* Blind derivation is the point: a non-blind run is not a weaker check,
|
|
2843
|
+
* it is no check. `false` is unrepresentable here and refused at runtime
|
|
2844
|
+
* for untyped callers.
|
|
2845
|
+
*/
|
|
2846
|
+
blind: true;
|
|
2847
|
+
}
|
|
2848
|
+
/** A validated, frozen spec. Produced only by `defineEquivalenceCheck`. */
|
|
2849
|
+
interface EquivalenceCheckDefinition {
|
|
2850
|
+
readonly spec: Readonly<EquivalenceCheckSpec>;
|
|
2851
|
+
}
|
|
2852
|
+
/** One arm's committed statement plus the provenance the record rests on. */
|
|
2853
|
+
interface EquivalenceArm {
|
|
2854
|
+
armId: string;
|
|
2855
|
+
/** The committed formal statement text (pilot: the arm's `Statement.lean` source). */
|
|
2856
|
+
statement: string;
|
|
2857
|
+
/**
|
|
2858
|
+
* What this arm derived the statement from. Pilot arm A: the paper's
|
|
2859
|
+
* LaTeX source; pilot arm B: the campaign's artifacts. The two arms
|
|
2860
|
+
* SHOULD derive from different renderings of the claim — that is what
|
|
2861
|
+
* makes agreement informative.
|
|
2862
|
+
*/
|
|
2863
|
+
derivedFrom: string;
|
|
2864
|
+
/**
|
|
2865
|
+
* The blindness declaration. Both flags must be `true`; a `false` value
|
|
2866
|
+
* is refused loudly, because recording a non-blind arm would launder an
|
|
2867
|
+
* invalid check into a valid-looking record.
|
|
2868
|
+
*/
|
|
2869
|
+
blindness: {
|
|
2870
|
+
/** This arm never saw any other arm's statement before committing its own. */
|
|
2871
|
+
toOtherArms: boolean;
|
|
2872
|
+
/** This arm never saw the verification outcome before committing. */
|
|
2873
|
+
toOutcome: boolean;
|
|
2874
|
+
};
|
|
2875
|
+
}
|
|
2876
|
+
type EquivalenceObligationStatus = 'proved' | 'refuted-with-separating-witness' | 'unresolved';
|
|
2877
|
+
/**
|
|
2878
|
+
* The discharged (or undischarged) equivalence obligation.
|
|
2879
|
+
*
|
|
2880
|
+
* Field presence is status-dependent and enforced by
|
|
2881
|
+
* `buildEquivalenceRecord`:
|
|
2882
|
+
* - `'proved'` requires `evidenceDigest` and forbids `separatingWitness`.
|
|
2883
|
+
* - `'refuted-with-separating-witness'` requires both the witness and
|
|
2884
|
+
* `evidenceDigest`.
|
|
2885
|
+
* - `'unresolved'` requires `unresolvedReason` and forbids the witness —
|
|
2886
|
+
* an undischarged obligation keeps its diagnostic instead of faking
|
|
2887
|
+
* either verdict.
|
|
2888
|
+
*/
|
|
2889
|
+
interface EquivalenceObligation {
|
|
2890
|
+
status: EquivalenceObligationStatus;
|
|
2891
|
+
/** The input on which the two statements provably disagree. */
|
|
2892
|
+
separatingWitness?: string;
|
|
2893
|
+
/** Why the obligation could not be discharged (checker error text, coverage limit). */
|
|
2894
|
+
unresolvedReason?: string;
|
|
2895
|
+
/** The checker that ran (or failed to run) the obligation. */
|
|
2896
|
+
checker: CheckerIdentity;
|
|
2897
|
+
/** Digest of the checker's evidence artifact (pilot: the kernel-checked `Check.lean` run). */
|
|
2898
|
+
evidenceDigest?: string;
|
|
2899
|
+
}
|
|
2900
|
+
/** The record the protocol produces — the by-hand pilot record, typed. */
|
|
2901
|
+
interface EquivalenceRecord {
|
|
2902
|
+
spec: Readonly<EquivalenceCheckSpec>;
|
|
2903
|
+
arms: readonly [EquivalenceArm, EquivalenceArm];
|
|
2904
|
+
obligation: EquivalenceObligation;
|
|
2905
|
+
}
|
|
2906
|
+
/** Input the bound checker receives: the artifact identity and both committed statements. */
|
|
2907
|
+
interface EquivalenceCheckerInput {
|
|
2908
|
+
artifact: string;
|
|
2909
|
+
statements: readonly [string, string];
|
|
2910
|
+
}
|
|
2911
|
+
/**
|
|
2912
|
+
* What a checker that RAN must return. `'unresolved'` is deliberately not
|
|
2913
|
+
* a checker result: a checker that could not decide returns
|
|
2914
|
+
* `{ succeeded: false, error }` through the port, and the protocol records
|
|
2915
|
+
* the unresolved obligation with that reason.
|
|
2916
|
+
*/
|
|
2917
|
+
interface EquivalenceCheckerResult {
|
|
2918
|
+
status: 'proved' | 'refuted-with-separating-witness';
|
|
2919
|
+
/** Required when `status` is refuted. */
|
|
2920
|
+
separatingWitness?: string;
|
|
2921
|
+
evidenceDigest: string;
|
|
2922
|
+
}
|
|
2923
|
+
/** An equivalence checker is a `StrategyChecker` — same port as every strategy. */
|
|
2924
|
+
type EquivalenceChecker = StrategyChecker<EquivalenceCheckerInput, EquivalenceCheckerResult>;
|
|
2925
|
+
/** A refused equivalence check. `code` names the exact refusal for programmatic handling. */
|
|
2926
|
+
declare class EquivalenceProtocolError extends Error {
|
|
2927
|
+
readonly code: 'arm-count' | 'not-blind' | 'arm-saw-other' | 'arm-saw-outcome' | 'duplicate-arm-id' | 'empty-field' | 'unknown-source' | 'witness-missing' | 'witness-on-proved' | 'witness-on-unresolved' | 'reason-missing' | 'evidence-missing' | 'checker-strategy-mismatch';
|
|
2928
|
+
constructor(code: EquivalenceProtocolError['code'], message: string);
|
|
2929
|
+
}
|
|
2930
|
+
/**
|
|
2931
|
+
* Validate and freeze a two-arm blind equivalence check spec.
|
|
2932
|
+
*
|
|
2933
|
+
* The literal types already refuse a wide design at compile time; the
|
|
2934
|
+
* runtime checks hold the same line for untyped callers. There is no
|
|
2935
|
+
* escape hatch: `arms: 3` or `blind: false` throws, never downgrades.
|
|
2936
|
+
*/
|
|
2937
|
+
declare function defineEquivalenceCheck(spec: EquivalenceCheckSpec): EquivalenceCheckDefinition;
|
|
2938
|
+
/**
|
|
2939
|
+
* Assemble the equivalence record, refusing every asymmetry.
|
|
2940
|
+
*
|
|
2941
|
+
* Refusals (each throws `EquivalenceProtocolError`):
|
|
2942
|
+
* - an arm whose `blindness.toOtherArms` is false — it saw the other
|
|
2943
|
+
* statement, so nothing was independently derived;
|
|
2944
|
+
* - an arm whose `blindness.toOutcome` is false — it could steer its
|
|
2945
|
+
* statement toward (or away from) the known result;
|
|
2946
|
+
* - duplicate arm ids, empty statements, empty derivations;
|
|
2947
|
+
* - an obligation whose fields contradict its status (see
|
|
2948
|
+
* `EquivalenceObligation`).
|
|
2949
|
+
*/
|
|
2950
|
+
declare function buildEquivalenceRecord(definition: EquivalenceCheckDefinition, arms: readonly [EquivalenceArm, EquivalenceArm], obligation: EquivalenceObligation): EquivalenceRecord;
|
|
2951
|
+
/**
|
|
2952
|
+
* Discharge the obligation through an injected checker and return the
|
|
2953
|
+
* record.
|
|
2954
|
+
*
|
|
2955
|
+
* Order matters: every arm refusal fires BEFORE the checker runs — an
|
|
2956
|
+
* invalid check must not spend. A checker whose `strategy` differs from
|
|
2957
|
+
* `spec.source` is refused for the same reason: a judge cannot silently
|
|
2958
|
+
* discharge a proof-kernel obligation.
|
|
2959
|
+
*
|
|
2960
|
+
* A checker failure (`succeeded: false`) is not thrown: it becomes an
|
|
2961
|
+
* `'unresolved'` obligation carrying the full error text, which is the
|
|
2962
|
+
* honest record of an undischarged check.
|
|
2963
|
+
*/
|
|
2964
|
+
declare function runEquivalenceCheck(definition: EquivalenceCheckDefinition, arms: readonly [EquivalenceArm, EquivalenceArm], checker: EquivalenceChecker): Promise<EquivalenceRecord>;
|
|
2965
|
+
//#endregion
|
|
3095
2966
|
//#region src/ui-finding.d.ts
|
|
3096
2967
|
/**
|
|
3097
2968
|
* UI audit finding — substrate primitive for "what is wrong with the UI?"
|
|
@@ -3761,73 +3632,6 @@ declare function promptBisect(options: {
|
|
|
3761
3632
|
offendingParagraphIndex?: number;
|
|
3762
3633
|
}>;
|
|
3763
3634
|
//#endregion
|
|
3764
|
-
//#region src/counterfactual.d.ts
|
|
3765
|
-
type CounterfactualMutation = {
|
|
3766
|
-
kind: 'swap-model';
|
|
3767
|
-
at: number;
|
|
3768
|
-
newModel: string;
|
|
3769
|
-
} | {
|
|
3770
|
-
kind: 'swap-tool-result';
|
|
3771
|
-
at: number;
|
|
3772
|
-
newResult: unknown;
|
|
3773
|
-
} | {
|
|
3774
|
-
kind: 'truncate-after';
|
|
3775
|
-
at: number;
|
|
3776
|
-
} | {
|
|
3777
|
-
kind: 'inject-system-message';
|
|
3778
|
-
at: number;
|
|
3779
|
-
content: string;
|
|
3780
|
-
} | {
|
|
3781
|
-
kind: 'custom';
|
|
3782
|
-
at: number;
|
|
3783
|
-
describe: string;
|
|
3784
|
-
apply: (step: TrajectoryStep) => TrajectoryStep;
|
|
3785
|
-
};
|
|
3786
|
-
interface CounterfactualContext {
|
|
3787
|
-
originalRunId: string;
|
|
3788
|
-
originalTrajectory: Trajectory;
|
|
3789
|
-
/** Steps up to (but not including) the mutation point — the prefix the
|
|
3790
|
-
* replayed agent inherits as its prior conversation/tool history. */
|
|
3791
|
-
prefix: TrajectoryStep[];
|
|
3792
|
-
mutation: CounterfactualMutation;
|
|
3793
|
-
/** Pre-applied mutation on the step at `mutation.at`. Consumers use this
|
|
3794
|
-
* as the FIRST step the replayed agent emits (they decide whether to
|
|
3795
|
-
* re-emit it or continue from there). */
|
|
3796
|
-
mutatedStep: TrajectoryStep;
|
|
3797
|
-
}
|
|
3798
|
-
interface CounterfactualResult {
|
|
3799
|
-
counterfactualRunId: string;
|
|
3800
|
-
originalRunId: string;
|
|
3801
|
-
mutation: CounterfactualMutation;
|
|
3802
|
-
/** Structured delta summary — caller can extend via scoring. */
|
|
3803
|
-
delta: {
|
|
3804
|
-
originalOutcomeScore: number | null;
|
|
3805
|
-
counterfactualOutcomeScore: number | null;
|
|
3806
|
-
deltaScore: number | null;
|
|
3807
|
-
};
|
|
3808
|
-
}
|
|
3809
|
-
interface CounterfactualRunner {
|
|
3810
|
-
/**
|
|
3811
|
-
* Execute the agent from `ctx.prefix` with the mutation applied.
|
|
3812
|
-
* MUST emit spans into the provided emitter so they become part of
|
|
3813
|
-
* the counterfactual run. MUST call emitter.endRun() with a verdict.
|
|
3814
|
-
*/
|
|
3815
|
-
executeFrom: (ctx: CounterfactualContext, emitter: TraceEmitter) => Promise<void>;
|
|
3816
|
-
}
|
|
3817
|
-
declare function runCounterfactual(store: TraceStore, originalRunId: string, mutation: CounterfactualMutation, runner: CounterfactualRunner): Promise<CounterfactualResult>;
|
|
3818
|
-
/**
|
|
3819
|
-
* Aggregate a batch of counterfactuals into a simple attribution table:
|
|
3820
|
-
* which mutation kinds move outcomes most? (Useful when you run a grid
|
|
3821
|
-
* over the same trajectory — swap-model at every llm span, swap-tool
|
|
3822
|
-
* at every tool span — and want a ranked summary.)
|
|
3823
|
-
*/
|
|
3824
|
-
declare function attributeCounterfactuals(results: CounterfactualResult[]): Array<{
|
|
3825
|
-
mutationKind: CounterfactualMutation['kind'];
|
|
3826
|
-
n: number;
|
|
3827
|
-
meanAbsDelta: number;
|
|
3828
|
-
meanSignedDelta: number;
|
|
3829
|
-
}>;
|
|
3830
|
-
//#endregion
|
|
3831
3635
|
//#region src/cross-trace-diff.d.ts
|
|
3832
3636
|
type AlignmentOp = {
|
|
3833
3637
|
op: 'match';
|
|
@@ -5558,14 +5362,20 @@ type SeatPresetName = keyof typeof seatPresets;
|
|
|
5558
5362
|
/**
|
|
5559
5363
|
* Tier presets — plain data, swap or spread freely.
|
|
5560
5364
|
*
|
|
5561
|
-
*
|
|
5562
|
-
*
|
|
5563
|
-
*
|
|
5564
|
-
*
|
|
5565
|
-
*
|
|
5566
|
-
*
|
|
5567
|
-
*
|
|
5568
|
-
*
|
|
5365
|
+
* A preset names REQUESTED ids, and a requested id is NOT a guarantee of
|
|
5366
|
+
* family, of provider, or of liveness. A routing gateway can accept any id
|
|
5367
|
+
* below and answer from a different model on HTTP 200, with only the response
|
|
5368
|
+
* body's `model` field betraying the swap. The served assertion is the
|
|
5369
|
+
* guarantee: gate a run on `assertModelsServed({ probe: true })` and assert
|
|
5370
|
+
* `assertServedModel` per call. `assertCrossFamily` over these ids proves the
|
|
5371
|
+
* configuration is diverse; only `assertCrossFamilyServed` over the ids that
|
|
5372
|
+
* ANSWERED proves the run was.
|
|
5373
|
+
*
|
|
5374
|
+
* `economy` names ids a live router probe answered from the provider their
|
|
5375
|
+
* name implies, so the judge trio spanned three provider families (deepseek /
|
|
5376
|
+
* zhipu / google) as configured and, at that probe, as served. A preset is
|
|
5377
|
+
* only as good as its last probe: ids go dead and start resolving elsewhere
|
|
5378
|
+
* without notice, so re-probe rather than trusting this list.
|
|
5569
5379
|
*
|
|
5570
5380
|
* `frontier` is deliberately EMPTY: entitled frontier ids vary per router
|
|
5571
5381
|
* account, and a hardcoded claude/gpt-5 id 401s on keys that lack it. Supply
|
|
@@ -5934,5 +5744,5 @@ type CachedJudge<TArtifact, TScenario extends Scenario$1 = Scenario$1> = JudgeCo
|
|
|
5934
5744
|
*/
|
|
5935
5745
|
declare function cachedJudge<TArtifact, TScenario extends Scenario$1 = Scenario$1>(judge: JudgeConfig$1<TArtifact, TScenario>, store: VerdictCacheStore, options: CachedJudgeOptions): CachedJudge<TArtifact, TScenario>;
|
|
5936
5746
|
//#endregion
|
|
5937
|
-
export { AGENT_PROFILE_KINDS, ATIF_SCHEMA_VERSION, ATTESTATION_ALGORITHM, type ActionExecutionPolicy, type ActionPolicyDecision, type ActionableSideInfo, type ActiveLearningOptions, type AdapterRun, AgentDriver, type AgentDriverConfig, AgentEvalError, type AgentEvalErrorCode, type AgentInterfaceProfileLike, type AgentProfile, type AgentProfileCell, type AgentProfileCellInput, type AgentProfileCellSchemaVersion, AgentProfileCellValidationError, type AgentProfileDimensionValue, type AgentProfileHarness, type AgentProfileJson, type AgentProfileJsonObject, type AgentProfileKind, type AgentProfileRuntimeReceipt, type AgentProfileSource, type AgentProfileSourceInput, type AlignmentOp, type Analyst, type AnalystContext, type AnalystCost, type AnalystFeedbackTrajectoryOptions, type AnalystFinding, type AnalystFindingDigest, type AnalystHooks, type AnalystInputKind, type AnalystMissedIssue, AnalystRegistry, type AnalystRegistryOptions, type AnalystRequirements, type AnalystReviewCounts, type AnalystReviewDecision, type AnalystReviewQuality, type AnalystReviewRequest, type AnalystReviewSource, type AnalystRunDigest, type AnalystRunEvent, type AnalystRunInputs, type AnalystRunResult, type AnalystRunSummary, type AnalystSeverity, type AnalystUsageReceipt, type AnalyzeTracesInput, type AnalyzeTracesOptions, type AnalyzeTracesResult, type AntiSlopConfig, type AntiSlopIssue, type AntiSlopReport, Artifact, type ArtifactCheck, type Artifact$1 as ArtifactCheckArtifact, type ArtifactEventLike, type ArtifactResult, type ArtifactValidator, type AsiSeverity, type AssertCapabilityHeadroomOptions, type AssertCrossFamilyOptions, type AssertCrossFamilyServedOptions, type AssertModelsServedOptions, type AssertServedModelOptions, type AssertSingleBackendOptions, type AttestationProvenance, type AttestationVerification, type AttestedReport, type AutoPrClient, BENCHMARK_SPLIT_SEED, BOOTSTRAP_GATE_MIN_N, type BackendDescriptor, BackendIntegrityError, type BackendIntegrityReport, type BaselineOptions, type BaselineReport, type BehaviorAssertion, type BehavioralMetrics, type BehavioralTokenSequence, type BenchmarkAdapter, type BenchmarkDatasetItem, type BenchmarkEvaluation, type BenchmarkFamily, type BenchmarkReport, type BenchmarkResponder, BenchmarkRunner, type BenchmarkRunnerConfig, type BenchmarkScenario, type BenchmarkSource, type BenchmarkTaskKind, type BisectOptions, type BisectResult, type BisectStep, type BlendWeights, type BootstrapOptions, type BootstrapResult, type BoundedTraceAnalysisStoreOptions, BudgetBreachError, BudgetGuard, BudgetLedgerEntry, type BudgetPolicy, BudgetSpec, type BuildTraceAnalysisToolsOptions, CODING_HARNESSES, CONTROL_INTEGRITY_ANALYST, type CachedJudge, type CachedJudgeOptions, type CalibrationResult, type CallExpectation, CallbackResearcher, type CallbackResearcherOptions, type CampaignFactoryParams, type CampaignIntegrityPolicy, type CampaignRunContext, type CampaignRunOutcome, type CampaignRunner, type CampaignScenario, type CampaignVariant, type CanaryAlert, type CanaryEvaluation, type CanaryKind, type CanaryLeak, type CanaryOptions, type CanaryReport, type CanarySeverity, type CandidateComparison, type CandidateScenario, type CandidateScore, type CapabilityHeadroomOptions, type CapabilityHeadroomResult, CaptureFetchContext, CaptureFetchOptions, CaptureIntegrityError, type CausalAttributionReport, type CellVerdict, type ChannelRollup, type ChatCallOpts, type ChatClient, type ChatMessage, type ChatRequest, type ChatResponse, type ChatToolCall, type ChatTransport, type CheckResult, type CliBridgeTransportOpts, type CliffsMagnitude, type ClusterBootstrapInterval, type ClusterSignFlipAlternative, type ClusterSignFlipResult, type ClusteredBinaryCluster, type ClusteredMatchedPair, type ClusteredPairedBinaryOptions, type ClusteredPairedBinaryResult, type ClusteredPairedBinaryStatistics, type CollectedArtifacts, type CommandRunner, type ComparePairedArmsOptions, type CompletionCriterion, type CompletionRequirement, type CompletionVerdict, type ConceptComplexity, type ConceptFinding, type ConceptSpec, type ConceptWeightStrategy, ConfigError, type ContinuityCheck, type ContinuityCheckResult, type ContinuityReport, type ContinuitySnapshotPair, type ContinuousAgreement, type ContinuousAgreementOptions, type ContinuousCalibrationResult, type ContractCheckResult, type ContractJudgeOptions, type ContractMetric, type ContractReport, type ContractRule, type ContractRuleKind, type ContractSpan, type ContractVerdict, type ContractViolation, type ControlActionFailureMode, type ControlActionOutcome, type ControlBudget, type ControlContext, type ControlDecision, type ControlEvalResult, ControlIntegrityAnalyst, type ControlRunResult, type ControlRunToRunRecordOptions, type ControlRuntimeConfig, type ControlRuntimeError, type ControlSeverity, type ControlStep, type ControlStopPolicies, ConvergenceTracker, type CorpusAgreementOptions, type CorpusAgreementPerDimension, type CorpusAgreementReport, type CorpusScoreRecord, type CorrectnessChecker, CostAccountingIncompleteError, CostCallConflictError, CostCeilingReachedError, type CostChannel, type CostEntry, CostLedger, type CostLedgerFilter, type CostLedgerHandle, type CostLedgerOptions, type CostLedgerPersistence, CostLedgerPersistenceError, type CostLedgerSummary, type CostProvenance, type CostReceipt, CostReceiptCaptureError, type CostReceiptInput, type CostReport, CostReservationExceededError, type CostResult, type CostSummary, CostTracker, type CostUsage, type CounterfactualContext, type CounterfactualMutation, type CounterfactualResult, type CounterfactualRunner, type CreateChatClientOpts, type CreateDefaultReviewerOptions, type CreateExperimentInput, type CreateSandboxPoolOpts, type CreateTraceAnalystOptions, CrossFamilyError, type CrossTraceDiff, type CrossTraceDiffOptions, type CustomTokenPricing, type CustomTransportOpts, DECISION_PAIRED_DELTA_STATISTIC, DEFAULT_AGENT_SLOS, DEFAULT_COMPLEXITY_WEIGHTS, DEFAULT_RULES as DEFAULT_FAILURE_RULES, DEFAULT_FINDERS, DEFAULT_HARNESS_OBJECTIVES, DEFAULT_MUTATION_PRIMITIVES, DEFAULT_MUTATORS, DEFAULT_PERMUTATIONS, DEFAULT_PR_REVIEW_SCORE_WEIGHTS, DEFAULT_REDACTION_RULES, DEFAULT_RED_TEAM_CORPUS, DEFAULT_RUN_SCORE_WEIGHTS, DEFAULT_SEVERITY_WEIGHTS, DEFAULT_TRACE_ANALYST_BUDGETS, DEFAULT_TRACE_ANALYST_KINDS, DEFAULT_TRACE_ANALYST_LIMITS, DataAcquisitionPlan, Dataset, type DatasetDifficulty, type DatasetManifest, type DatasetOverview, type DatasetProvenance, type DatasetScenario, type DatasetSplit, type DecideNextUserTurnOpts, type DefaultAnalystRegistryOptions, type DefaultVerdict, type DeltaStatistic, type DeployFamily, type DeployGateLayerInput, type DeployRunResult, type DeployRunner, type DescriptionLengthCandidate, type DescriptionLengthConfig, type DescriptionLengthDecision, type DescriptionLengthEvidence, DescriptionLengthGate, type DescriptionLengthRejectionCode, type DetectorEvent, type DetectorSeverity, type DetectorSignal, type DiffPolicy, type DiffScorecardOptions, type DirEntry, type DirectProviderTransportOpts, type Direction, type DiscoverPersonasOptions, type DiscoveredPersona, DockerSandboxDriver, type DriverResult, type DriverState, type DspyRlmTraceEngineOptions, DualAgentBench, type DualAgentBenchConfig, type DualAgentReport, type DualAgentRound, type DualAgentScenario, type DualAgentScenarioResult, type EProcess, type EProcessOptions, type EProcessState, type EProcessStep, ERROR_COUNT_PATTERNS, type EnsembleAggregate, type EnsembleJudgeOptions, type ErrorCluster, type ErrorCountPattern, type ErrorStreakOptions, type EvalCampaignOptions, type EvalCampaignResult, type EvalResult, type EvalToolDef, EvalTraceStore, EventFilter, EventKind, type EvidenceRef, type EvolutionRound, type ExactAnalystBudgetPolicy, type ExactAnalystBudgetSnapshot, type ExactAnalystExecutionPlanSnapshot, type ExactAnalystRunCompletion, type ExactAnalystRunEvent, ExactAnalystRunExecutionError, type ExactAnalystRunPolicySnapshot, type ExactAnalystRunResult, type ExactAnalystRunSummary, type ExactAnalystSnapshot, type ExactCapableAnalyst, type ExactExecutionComponentIdentity, type ExactExecutionComponentSnapshot, type ExactRegistryRunOpts, type ExactRiskDifferenceResult, type ExecutorConfig, type Expectation, type Experiment, type ExperimentPlan, type ExperimentProvenance, type ExperimentRep, type ExperimentResult, type ExperimentStats, type ExperimentStore, ExperimentTracker, type ExperimentTrackerOptions, type ExperimentVerdict, ExportableSpan, type ExportedRewardModel, type ExtractOptions, type ExtractResult, ExtractUsageFromSseOptions, ExtractedUsage, FAILURE_CLASSES, FAILURE_MODE_KIND_SPEC, type FactorContribution, type FactorialCell, type FailedRun, FailureClass, type FailureClassification, type FailureContext, type FailureMode, type FailureRule, type FeedbackArtifactType, type FeedbackAttempt, type FeedbackLabel, type FeedbackLabelKind, type FeedbackLabelSource, type FeedbackOptimizerRow, type FeedbackOutcome, type FeedbackPattern, type FeedbackReplayAdapter, type FeedbackReplayResult, type FeedbackSeverity, type FeedbackSplitPolicy, type FeedbackTask, type FeedbackTrajectory, type FeedbackTrajectoryFilter, type FeedbackTrajectoryStore, type FieldDestination, type FileChange, FileSystemFeedbackTrajectoryStore, FileSystemRawProviderSink, FileSystemRawProviderSinkOptions, FileSystemTraceStore, FileSystemTraceStoreOptions, type Finding, type FindingSubject, type FindingSubjectKind, type FindingsDiff, FindingsStore, type FlattenOtlpOptions, type FlowAction, type FlowLayerEnv, type FlowLayerFactoryInput, type FlowRunner, type FlowRunnerStepResult, type FlowSpec, type FlowStep, type FromHarborOptions, type GainDistributionBin, type GainDistributionFigureSpec, type GainDistributionOptions, type GateDecision, type GateEvidence, GenericSpan, type GhCliClientOptions, type GoldenItem, type GoldenSeverity, type GoldenSpec, HARBOR_IMPORT_GAP, HARNESS_BRIEFS, HARNESS_NATIVE_MODEL, type HarborAgent, type HarborContentPart, type HarborFinalMetrics, type HarborImageSource, type HarborMetrics, type HarborObservation, type HarborObservationResult, type HarborStep, type HarborStepSource, type HarborSubagentTrajectoryRef, type HarborToolCall, type HarborTrajectory, type HarnessAdapter, type HarnessConfig, type HarnessExperimentConfig, type HarnessExperimentResult, type HarnessIntervention, type HarnessRunRequest, type HarnessRunResult, type HarnessScenario, type HarnessSelection, type HarnessType, type HarnessVariant, type HarnessVariantReport, type HeadroomClass, type HeadroomInput, HeldOutGate, type HeldOutGateConfig, type HeldOutGateRejectionCode, type HeldOutPartition, type HiddenCriteriaGrader, type HiddenGradeResult, type HiddenLeak, HoldoutAuditor, HoldoutLockedError, type HttpGithubClientOptions, type HypothesisManifest, type HypothesisResult, IMPROVEMENT_KIND_SPEC, INPUT_VALUE, INTENT_MATCH_JUDGE_VERSION, type ImageData, type ImprovementThresholds, type ImprovementVerdictResult, InMemoryFeedbackTrajectoryStore, InMemoryRawProviderSink, InMemoryRawProviderSinkOptions, InMemoryTraceStore, InMemoryWorkspaceInspector, type InferenceScorer, type InspectorContext, type IntentMatchInput, type IntentMatchOptions, type IntentMatchResult, type InteractionContribution, type InterimReleaseConfidence, type InterimReleaseConfidenceInput, type JudgeConfig, JudgeError, type JudgeFamily, type JudgeFleetOptions, type JudgeFn, type JudgeInput, JudgeParseError, type JudgeReplayGateArgs, type JudgeReplayResult, type JudgeRetryOutcome, type JudgeRetryPolicy, type JudgeRubric, JudgeRunner, type JudgeScore, type JudgeScoreInput, type JudgeScoresRecord, JudgeSpan, type JudgeVerdict, KNOWLEDGE_GAP_KIND_SPEC, KNOWLEDGE_POISONING_KIND_SPEC, type KeywordConceptSpec, type KeywordCoverageFinding, type KeywordCoverageOptions, type KeywordCoverageResult, KnowledgeAcquisitionMode, KnowledgeBundle, KnowledgeFallbackPolicy, KnowledgeFreshness, KnowledgeImportance, KnowledgeReadinessReport, KnowledgeRecommendedAction, KnowledgeRequirement, KnowledgeRequirementCategory, KnowledgeResponsibleSurface, KnowledgeSensitivity, LLM_CACHED_TOKENS, LLM_CACHED_TOKEN_ATTR_KEYS, LLM_CACHE_WRITE_TOKENS, LLM_CACHE_WRITE_TOKEN_ATTR_KEYS, LLM_CONTEXT_TOKENS, LLM_COST_ATTR_KEYS, LLM_COST_USD, LLM_INPUT_TOKENS, LLM_INPUT_TOKEN_ATTR_KEYS, LLM_MODEL_ATTR_KEYS, LLM_MODEL_NAME, LLM_OUTPUT_TOKENS, LLM_OUTPUT_TOKEN_ATTR_KEYS, LLM_REASONING_TOKENS, LLM_REASONING_TOKEN_ATTR_KEYS, type LangfuseEnvelope, type LangfuseGeneration, type LangfuseScore, type Layer, type LayerResult, type LayerStatus, type LeaderboardOptions, type LeaderboardRow, LimitExceededError, type LiveProofArtifact, type LiveProofConfig, type LiveProofContext, type LiveProofResult, LlmCallError, type LlmCallMetadata, type LlmCallRequest, type LlmCallResult, LlmClient, type LlmClientOptions, type LlmCorrectnessCheckerOpts, type LlmJsonCall, type LlmJudgeDimension, type LlmJudgeOptions, type LlmMessage, LlmResponseError, type LlmReviewerConfig, LlmRouteAssertionError, type LlmRouteRequirements, LlmSpan, LlmSpanOtlpInput, type LlmUsage, LockedJsonlAppender, MANN_WHITNEY_EXACT_MAX_STATES, MANN_WHITNEY_EXACT_MAX_WORK, MODEL_PRICING, type MakeEvalToolsConfig, type MannWhitneyResult, type MatchResult, type MatchedPair, type MatchedRunRecordPair, type MatcherResult, type MaximumCharge, type McNemarResult, type Measured, type MeasurementPolicy, type MergeOptions, Message, type MetricSamples, type MetricVerdict, MetricsCollector, type MintRolloutOptions, type MintRolloutResult, type MintedRolloutLine, type MintedRolloutOutcome, type MockTransportOpts, type ModelCostRollup, type ModelPreflight, type ModelSeats, ModelSubstitutionError, ModelsUnreachableError, type MuffledFinder, type MuffledFinding, MultiLayerVerifier, type MultiToolchainLayerConfig, type Mutator, Mutex, type NoLeakOptions, type NoProgressOptions, NoopRawProviderSink, NoopResearcher, NotFoundError, OPENINFERENCE_SPAN_KIND, OTEL_AGENT_EVAL_SCOPE, OUTPUT_VALUE, type Objective, type Oracle, type OracleObservation, type OracleReport, type OracleResult, type OrthogonalityInput, type OrthogonalityResult, OtelExportConfig, OtelExporter, type OtelPipelineHandle, type OtelPipelineOptions, OtlpExport, OtlpFileTraceStore, type OtlpFileTraceStoreOptions, type OtlpFlatLine, OtlpResourceSpans, OtlpSpan, OtlpSpanRole, OtlpSpanRoleInput, type OtlpToRunRecordsOptions, type OtlpTraceRunRecord, type PaidCallResult, type PairArmsOptions, type PairArmsResult, type PairRunRecordsResult, type PairedArmRow, type PairedArmsComparison, type PairedBootstrapOptions, type PairedBootstrapResult, type PairedCorrectness, type PairedDecisionMethod, type PairedDecisionShape, type PairedDecisionStatistic, type PairedDeltaTestOptions, type PairedDeltaTestResult, type PairedEvalueOptions, type PairedEvalueSequence, type PairedEvalueStep, type PairedMcNemarEvidence, type PairedMetricDelta, type PairedPromotionDecision, type PairedPromotionDecisionOptions, type PairedSignTestResult, type PairedTTestResult, PairwiseSteeringOptimizer, type ParaphraseRobustnessScenarioInput, type ParaphraseRobustnessScenarioResult, type ParetoFigureSpec, type ParetoPoint, type ParetoResult, type PartitionHeldOutOptions, type PendingCostCall, type PendingCostCallView, type PersistedFinding, type PersonaConfig, type PersonaRigor, type Playbook, type PlaybookEntry, type PoolSlot, type PositionalBiasResult, type PrReviewAuditCase, type PrReviewBenchmarkSummary, type PrReviewComment, type PrReviewMatchedFinding, type PrReviewOutcome, type PrReviewReferenceFinding, type PrReviewScore, type PrReviewScoreWeights, type PrReviewSeverity, type PrReviewSource, type PreferenceMemoryEntry, type PreflightModelsOptions, type PreflightOutcome, type ProducedProposal, type ProducedState, type ProductBenchmarkArm, type ProductBenchmarkArtifactPaths, type ProductBenchmarkBudgets, type ProductBenchmarkExportOptions, type ProductBenchmarkExportResult, type ProductBenchmarkManifest, type ProductBenchmarkProfileRef, type ProductBenchmarkRecord, type ProductBenchmarkRepoRef, type ProductBenchmarkRunInput, type ProductBenchmarkScenario, type ProductBenchmarkSingleRunExportOptions, type ProductBenchmarkSplit, type ProductBenchmarkSubstrateVersions, type ProductBenchmarkValidationReport, ProductClient, type ProductClientConfig, type ProfileAxisSpec, type ProjectRuntimeTrajectoryEvidenceOptions, type ProjectedOtlpSpan, type PromptHandle, PromptRegistry, type ProportionInterval, type ProposalEventLike, type ProposalFinding, type ProposalFindingOrigin, type ProposeAutomatedPullRequestInput, type ProposeAutomatedPullRequestResult, type ProposeFn, type ProposeInput, type ProposeOutput, type ProposeReviewConfig, type ProposeReviewControlAction, type ProposeReviewControlConfig, type ProposeReviewControlResult, type ProposeReviewControlState, type ProposeReviewReport, type ProposeReviewShot, type ProposedSideEffect, type ProvenanceReader, ProviderRedactor, type QueryTracesPage, REDACTION_VERSION, REFERENCE_EQUIVALENCE_INPUT_LIMITS, REFERENCE_EQUIVALENCE_JUDGE_VERSION, RESEARCH_REPORT_HARD_PAIR_FLOOR, ROLLOUT_SCHEMA, RUN_COST_ATTR_KEYS, type RankTestMethod, type RankTestMethodRequest, type RankTestOptions, type RawAnalystEvidence, type RawAnalystFinding, RawProviderDirection, RawProviderEvent, RawProviderSink, RawProviderSinkFilter, type RecordRunsOptions, type RedTeamCase, type RedTeamCategory, type RedTeamFinding, type RedTeamPayload, type RedTeamReport, RedactionReport, RedactionRule, type ReferenceEquivalenceJudgeInput, type ReferenceEquivalenceJudgeOptions, type ReferenceEquivalenceJudgeResult, type ReferenceEquivalenceScenario, type ReferenceMatchResult, type ReferenceReplayAdapter, type ReferenceReplayAdapterFn, type ReferenceReplayAdapterLike, type ReferenceReplayAggregate, type ReferenceReplayCandidate, type ReferenceReplayCase, type ReferenceReplayCaseRun, type ReferenceReplayExecutionScenario, type ReferenceReplayItem, type ReferenceReplayMatch, type ReferenceReplayMatchStrategy, type ReferenceReplayMatcher, type ReferenceReplayPromotionDecision, type ReferenceReplayPromotionPolicy, type ReferenceReplayRun, type ReferenceReplayRunContext, type ReferenceReplayRunOptions, type ReferenceReplayRunStore, type ReferenceReplayScenario, type ReferenceReplayScenarioScore, type ReferenceReplayScore, type ReferenceReplayScoreOptions, type ReferenceReplaySplit, type ReferenceReplaySplitComparison, type ReferenceReplaySteeringRowsOptions, type ReflectionContext, type ReflectionProposal, type RegistryRunOpts, type ReleaseConfidenceAxis, type ReleaseConfidenceAxisName, type ReleaseConfidenceInput, type ReleaseConfidenceIssue, type ReleaseConfidenceMetrics, type ReleaseConfidenceScorecard, type ReleaseConfidenceStatus, type ReleaseConfidenceThresholds, type ReleaseTraceEvidence, type RenderReleaseReportOptions, type RepeatedActionOptions, ReplayCache, type ReplayCacheEntry, ReplayCacheMissError, type ReplayCacheStats, ReplayError, type ReplayFetchOptions, type RepoRef, type RequirementCheck, type ResearchReport, type ResearchReportCandidate, type ResearchReportDecision, type ResearchReportMethodology, type ResearchReportOptions, type ResearchReportRecommendation, type Researcher, RetrievalSpan, type Review, type ReviewFn, type ReviewInput, type ReviewMemoryEntry, type ReviewMemoryStore, type ReviewerMemoryEntry, type ReviewerOutput, type ReviewerPromptInput, type ReviewerSoftFailDefaults, type ReviewerVerificationSummary, type RewardRow, type RiskDifferenceResult, type RobustnessResult, type RolloutCapture, type RolloutLine, type RolloutRole, type RolloutScrubber, type RolloutSplit, type RolloutStep, type RouteMap, type RoutedField, type RouterTransportOpts, type RubricDimension, Run, type RunCommandInput, type RunCommandResult, RunCompleteHook, RunCompleteHookContext, type RunCostProvenance, RunCritic, type RunCriticOptions, type RunEvidenceMetadata, RunFilter, RunIntegrityError, RunIntegrityExpectations, RunIntegrityIssue, RunIntegrityIssueCode, RunIntegrityReport, type RunJudgeMetadata, RunLayer, RunOutcome, type RunPaidCallInput, type RunRecord, type RunRecordBackend, type RunRecordFilter, RunRecordValidationError, type RunScore, type RunScoreWeights, type RunSplitTag, RunStatus, type RunTaskFailure, type RunTerminalOutcome, type RunTokenUsage, type RunTrace, type RuntimeEventLike, type RuntimeResolution, type RuntimeTrajectoryEvidenceProjection, type RuntimeTrajectoryEvidenceSummary, type RuntimeTrajectoryHookEvent, type RuntimeTrajectoryRecord, type RuntimeTrajectoryRunRecord, SEMANTIC_CONCEPT_JUDGE_VERSION, SKILL_USAGE_ANALYST, SPAN_KIND_ATTR_KEYS, SUPERVISOR_RUN_INTEGRITY_SCHEMA, SUPERVISOR_RUN_SCHEMA, type SandboxDriver, SandboxHarness, type SandboxHarnessResult, type SandboxJudgeKind, type SandboxJudgeResult, type SandboxJudgeSpec, type SandboxPool, type SandboxResult, type SandboxSdkTransportOpts, SandboxSpan, type SatisfiedBy, type ScanOptions, type Scenario, type ScenarioCost, type ScenarioFile, ScenarioRegistry, type ScenarioResult, ScoreKnowledgeReadinessOptions, type ScoreOrigin, type ScorePreference, type ScoreRiskDifferenceResult, type Scorecard, type ScorecardCell, type ScorecardCellDiff, type ScorecardDiff, type ScorecardEntry, type ScorecardLogLine, type ScoredTarget, type SearchSpanResult, type SearchTraceResult, type SeatName, type SeatPresetName, SeatUnsetError, type SelfPlayOptions, type SelfPlayProposer, type SelfPlayScorer, type SelfPreferenceResult, type SemanticConceptJudgeInput, type SemanticConceptJudgeOptions, type SemanticConceptJudgeResult, type SequentialDecision, type SerializedRegex, type SeriesConvergenceOptions, type SeriesConvergenceResult, ServedCrossFamilyError, type ServedModelCheck, type ServedModelVerdict, type Severity, type SftExportOptions, type SftRow, type SignTestAlternative, type SignedManifest, type SignedManifestAlgo, type SingleBackendDivergence, SingleBackendError, type SingleBackendField, type SingleBackendReport, SkillUsageAnalyst, type SliceOptions, type Slo, type SloCheckResult, type SloComparator, type SloReport, type SloSeverity, type SlopCategory, type SlotFactory, type SourceLimits, Span, SpanBase, SpanFilter, SpanHandle, SpanKind, type SpanMatchRecord, SpanNotFoundError, type SpanPredicate, SpanStatus, type SplitCoverage, SseUsageMode, type SteeringBundle, type SteeringChange, type SteeringDelta, type SteeringOptimizationResult, type SteeringOptimizationRow, type SteeringOptimizationSelector, type SteeringOptimizerBackend, type SteeringOptimizerConfig, type SteeringRolePrompt, type StepAttribution, type StopDecision, type StreamingDetector, type SuboptimalCode, type SuboptimalSignal, SubprocessSandboxDriver, type SubprocessSandboxDriverOptions, type SummaryTable, type SummaryTableOptions, type SummaryTableRow, type SupervisorRunIntegrityEvidence, type SupervisorRunIntegrityIssue, type SupervisorRunIntegrityIssueCode, type SupervisorRunIntegrityOptions, type SupervisorRunIntegrityReport, type SupervisorRunIntegritySeverity, type SupervisorRunNodeRole, type SupervisorRunReader, type SupervisorRunReport, type SupervisorRunRollup, type SupervisorRunSourceOnlyCheckCode, type SupervisorRunSources, type SupervisorRunTree, type SupervisorRunTreeGap, type SupervisorRunTreeGapCode, type SynthesisReason, type SynthesisTarget, TOOL_ARGS_CAPTURED, TOOL_LATENCY_MS, TOOL_NAME, TOOL_NAME_ATTR_KEYS, TRACE_ANALYSIS_LIMITS, TRACE_ANALYST_ACTOR_DESCRIPTION, TRACE_ANALYST_ACTOR_DESCRIPTION_VERSION, TRACE_ANALYST_TOOL_NAMESPACE, TRACE_ANALYST_TRUNCATION_MARKER_PREFIX, TRACE_SCHEMA_VERSION, type TaskGold, type TaskHeadroom, type TestGradedRunOptions, type TestGradedRunResult, type TestGradedScenario, type TestOutputParser, type TestResult, type TextMatcher, type ThresholdContract, TokenCounter, type TokenSpec, type ToolCallEventLike, type ToolDef, type ToolMatcher, ToolSpan, ToolSpanOtlpInput, type ToolSpansToTraceAnalysisStoreOptions, type ToolStats, ToolTraceMissingError, type ToolUseMetrics, type ToolUseOptions, type TraceAggregate, type TraceAnalysisEngine, type TraceAnalysisEngineRequest, type TraceAnalysisEngineResult, TraceAnalysisLimitError, type TraceAnalysisStore, type TraceAnalysisStoreContext, TraceAnalysisStoreContractError, type TraceAnalysisToolDescriptor, TraceAnalysisValidationError, type TraceAnalystByteBudgets, type TraceAnalystDefinition, type TraceAnalystFilters, type TraceAnalystHookOptions, type TraceAnalystLimits, type TraceAnalystSpan, type TraceAnalystSpanKind, type TraceAnalystSpanStatus, type TraceAnalystTraceSummary, type TraceContract, TraceContractBuilder, TraceEmitter, TraceEmitterOptions, TraceEvent, TraceFileMalformedError, TraceFileMissingError, TraceFileTooLargeError, type TraceInsightContext, type TraceInsightFinding, type TraceInsightPanelRole, type TraceInsightPromptInput, type TraceInsightQualityGate, type TraceInsightQuestion, type TraceInsightReadiness, type TraceInsightSuite, type TraceInsightTask, TraceNotFoundError, TraceStore, TraceStoreSource, TraceStoreToOtlpOptions, type TracedAnalystOptions, type TracedJudgeOptions, TracesToOtlpResult, type Trajectory, type TrajectoryStep, type TreatmentClass, type TreatmentGate, type TreatmentGateInput, type TreatmentGateOptions, type TrialTrace, type Turn, type TurnMetrics, type TurnResult, UI_FINDING_SEVERITIES, UI_LENSES, UNIVERSAL_FINDERS, type UiFinding, type UiFindingScreenshot, type UiFindingSeverity, type UiLens, type Unavailable, UserQuestion, type ValidationContext, ValidationError, type ValidationIssue, type ValidationResult, type VerbosityBiasResult, type Verdict, type VerdictCacheStats, type VerdictCacheStore, type Verification, VerificationError, type VerificationReport, type VerifyContext, type VerifyFn, type VerifyOptions, type ViewSpansResult, type ViewTraceOversized, type ViewTraceResult, type VisualDiffOptions, type VisualDiffResult, type ViteDeployRunnerInput, WILCOXON_EXACT_MAX_N, WORKER_DRIVER_DOCTRINE, type WeightedCompositeInput, type WeightedCompositeResult, type WelchTestResult, type WelchTestStatus, type WilcoxonSignedRankResult, type WorkflowTopology, type WorkspaceAssertion, type WorkspaceAssertionResult, type WorkspaceInspector, type WorkspaceSnapshot, type WranglerDeployRunnerInput, acquisitionPlansForKnowledgeGaps, agentProfileCellHashMaterial, agentProfileCellKey, agentProfileHash, agentProfileId, agentProfileModelId, agentVisibleFields, aggregateJudgeVerdicts, aggregateLlm, aggregatePrReviewScore, aggregateRunScore, allCriticalPassed, analystFindingDigest, analystRunDigest, analystRunToFeedbackTrajectory, analystRunToReviewRequests, analyzeAntiSlop, analyzeSeries, analyzeSupervisorRun, analyzeSupervisorRunIntegrity, analyzeSupervisorRunSources, analyzeTraces, appendScorecard, applyLlmSpanOtlpAttributes, applyToolSpanOtlpAttributes, argHash, asNumber, asString, assertCapabilityHeadroom, assertCrossFamily, assertCrossFamilyServed, assertExactRegistryRunOpts, assertLlmRoute, assertMinted, assertMintedLines, assertModelsServed, assertNoHiddenLeak, assertProductBenchmarkRun, assertRealAgentReceipts, assertRealBackend, assertReleaseConfidence, assertRolloutLine, assertRunAgentProfileCell, assertRunCaptured, assertServedModel, assertServedModels, assertSingleBackend, assignFeedbackSplit, assignHeldOutTag, attachCostToReport, attest, attributeCounterfactuals, backoffMs, deterministicSplit as benchmarkDeterministicSplit, index_d_exports as benchmarks, benjaminiHochberg, bisect, blendHeldout, blockingKnowledgeEval, bonferroni, bootstrapCi, buildAgentInterfaceProfileCell, buildAgentProfileCell, buildDefaultAnalystRegistry, buildDriverSystemPrompt, buildProductBenchmarkManifest, buildReflectionPrompt, buildReviewerPrompt, buildTraceAnalysisToolDescriptors, buildTraceInsightContext, buildTraceInsightPrompt, buildTrajectory, byteLengthRange, cachedJudge, calibrateJudge, calibrateJudgeContinuous, callLlm, callLlmJson, canaryLeakView, canonicalJson, canonicalize, capabilityHeadroom, captureFetchToRawSink, causalAttribution, checkBehavioralCanary, checkCanaries, checkServedModel, checkSlos, checkTraceContracts, clamp01, classifyFailure, classifyOtlpSpanRole, classifyTreatment, claudeCodeSupervisorRunReader, cliffsDelta, clusteredPairedBinary, cohensD, collectionPreserved, commentsForSource, commitBisect, comparePairedArms, compareReferenceReplay, compareToBaseline, compilerJudge, completionVerdict, composeParsers, composeValidators, computeExperimentStats, computeFindingId, computeToolUseMetrics, computeTraceMetrics, confidenceInterval, containsAll, contentHash, contextInputTokens, continuousAgreement, contractJudge, controlFailureClassFromVerification, controlRunToFeedbackTrajectory, controlRunToRunRecord, convertTraceStoresToOtlp, corpusInterRaterAgreement, corpusInterRaterAgreementFromJudgeScores, costForTokenPricing, costForUsage, costReceiptFromLlm, costReceiptFromLlmError, costReport, createAntiSlopJudge, createBoundedTraceAnalysisStore, createChatClient, createDefaultReviewer, createDspyRlmTraceEngine, createFeedbackTrajectory, createIntentMatchJudge, createLlmCorrectnessChecker, createLlmReviewer, createOtelExporter, createOtelTracingStore, createReferenceEquivalenceJudge, createReplayFetch, createSandboxPool, createSemanticConceptJudge, createTokenRecallChecker, createTraceAnalyst, crossTraceDiff, crowdingDistance, dataDescriptionBits, decideNextUserTurn, decidePairedPromotion, decideReferenceReplayPromotion, decideReferenceReplayRunPromotion, defaultBlendWeights, defaultIsMaterial, defaultProviderRedactor, defaultReferenceReplayMatcher, defaultTraceInsightPanel, defineTraceAnalyst, deployGateLayer, describeTraceInsightScope, diffFindings, diffScorecard, discoverPersonas, distillPlaybook, domainEvidencePattern, dominates, eProcess, emitControlIntegrityFindings, ensembleJudge, errorStreakDetector, estimateCost, estimateTokens, evaluateActionPolicy, evaluateContract, evaluateHypothesis, evaluateInterimReleaseConfidence, evaluateOracles, evaluateReleaseConfidence, evaluateTraceContract, executeScenario, expandProfileAxes, expectAgent, exportProductBenchmark, exportProductBenchmarkRuns, exportRewardModel, exportRunAsOtlp, extractAssetUrls, extractErrorCount, extractOtlpAttributes, extractProducedState, extractUsage, extractUsageFromResponse, extractUsageFromSse, feedbackTrajectoriesToDatasetScenarios, feedbackTrajectoriesToOptimizerRows, feedbackTrajectoryToDatasetScenario, feedbackTrajectoryToOptimizerRow, fileContains, fileExists, fileExperimentStore, fileVerdictCache, findAutoMatchNoExpectation, findConstructorCwdDropped, findFallbackToPass, findLiteralTruePass, findProductBenchmarkArtifacts, findSkipCountsAsPass, firstNumberAttr, firstStringAttr, flattenOtlpExportToNdjson, flowLayer, fnv1a32, formatBenchmarkReport, formatDriverReport, formatFindings, formatScorecardDiff, fromHarborTrajectory, gainHistogram, gateTreatmentApplied, gateTreatmentFromMetrics, gateTreatmentFromSpans, gateTreatmentFromToolSpans, ghCliClient, gitProvenanceReader, precision as goldenPrecision, gradeOnHidden, gradeSemanticStatus, groupBy, groupRunsByAgentProfileCell, harnessAxisOf, hasCapturedToolArgs, hashContent, hashJson, hashScenarios, hashToUnit, hiddenGrade, holm, htmlContainsElement, httpGithubClient, improvementVerdict, inMemoryExperimentStore, inMemoryReferenceReplayStore, inMemoryReviewStore, inMemoryRunRecordBackend, inMemoryVerdictCache, inferDomainKeywords, inferOtlpKind, interRaterReliability, interpretCliffs, iqr, isBinaryOutcomeVector, isHiddenDestination, isJudgeSpan, isLlmSpan, isModelPriced, isOtelConfigured, isOtlpModelCall, isRealnessGated, isRetrievalSpan, isRolloutLine, isRunRecord, isRuntimeSupervisorRunDir, isSandboxSpan, isToolSpan, isTrainableSplit, isTransientLlmError, isUnavailable, iterateRawCalls, jestTestParser, jsonHasKeys, jsonShape, jsonlReferenceReplayStore, jsonlReviewStore, jsonlRunRecordBackend, judgeFamily, judgeReplayGate, judgeSpans, keyPreserved, knowledgeReadinessTracePayload, leaderboard, linterJudge, llmJudge, llmSpanFromProvider, llmSpans, loadScorecard, loadScorerFromGrader, localCommandRunner, lowercaseMutator, makeEvalTools, makeFinding, makeProposalFinding, mannWhitneyU, mapConcurrent, matchGoldens, matchSpan, maximumChargeForLlmRequest, mcnemar, mcnemarPower, mcnemarRequiredN, mergeLayerResults, mergeSteeringBundle, minimumPairsForPairedDeltaTest, mintRolloutRows, modelDescriptionBits, modelHasSnapshot, modelPriceKey, mulberry32, multiToolchainLayer, noProgressDetector, normalCdf, normalizeModelId, normalizeScores, notBlocked, objectiveEval, observeAll, observedScore, observedSplitScore, otelRunCompleteHook, otlpRowsToRunRecords, otlpRowsToTraceRunRecords, otlpTextToTraceAnalysisStore, otlpToRunRecords, otlpToTraceRunRecords, pairArms, pairRunRecords, pairedBinaryScale, pairedBootstrap, pairedCohensDz, pairedDecisionShape, pairedDeltaTest, pairedDeltaTieFraction, pairedEvalueSequence, pairedMde, pairedRiskDifference, pairedRiskDifferenceExact, pairedRiskDifferenceScore, pairedSignTest, pairedTTest, paraphraseRobustness, paraphraseRobustnessScenarios, paretoChart, paretoFrontier, paretoFrontierWithCrowding, parseCorrectnessResponse, parseFeedbackTrajectoriesJsonl, parseReflectionResponse, parseRunRecordSafe, parseRuntimeTrajectoryHookEvent, partialCredit, partitionHeldOut, passAtK, passOrthogonality, pearsonR, pixelDeltaRatio, planTraceInsightQuestions, politenessPrefixMutator, positionalBias, preflightModels, printDriverSummary, probeLlm, productBenchmarkIntegrityFailures, productBenchmarkMutableSurfaces, productBenchmarkRepoIdentity, productBenchmarkSplits, index_d_exports$1 as profile, projectOtlpFlatLine, projectRuntimeTrajectoryEvidence, promptBisect, proposeSynthesisTargets, providerFromBaseUrl, pytestTestParser, ranks, readClaudeCodeSupervisorRun, readOtlpStatus, readProductBenchmarkManifest, readProductBenchmarkRecords, readRuntimeSupervisorRun, recordRuns, recordRunsToScorecard, redTeamDataset, redTeamReport, redactString, redactValue, referenceReplayRunsToSteeringRows, referenceReplayScenarioToRunScore, regexMatch, regexMatches, relabelImportedSplit, renderMarkdownReport, renderPlaybookMarkdown, renderPreferenceMemoryMarkdown, renderPriorFindings, renderReleaseReport, renderSteeringText, renderSupervisorRunHeadline, renderSupervisorRunMarkdown, renderUpstreamFindings, repeatedActionDetector, replayFeedbackTrajectories, replayFeedbackTrajectory, replayScorerOverCorpus, replayTraceThroughJudge, requireAgentProfileCell, requiredPairedSampleSize, requiredSampleSize, researchReport, resetDeprecationWarnings, resolveModelPricing, resolveSeat, resolveTraceAnalystLimits, rollupSupervisorRuns, roundTripRunRecord, routeFields, rowCount, rowWhere, runAgentControlLoop, runAssertions, runBehavioralCanaries, runCanaries, runCounterfactual, runE2EWorkflow, runEvalCampaign, runExpectations, runFailureClass, runHarnessExperiment, runIntentMatchJudge, runJudgeFleet, runKeywordCoverageJudge, runKeywordCoverageJudgeUrl, runLiveProof, runProposeReview, runProposeReviewAsControlLoop, runRecordToProductBenchmarkRecord, runReferenceEquivalenceJudge, runReferenceReplay, runScore, runSelfPlay, runSemanticConceptJudge, runTaskScore, runTestGradedScenario, runTraceAnalyst, runsForScenario, runtimeSupervisorRunReader, scalarScore, scanForMuffledGates, scoreContinuity, scoreFromEvals, scoreKnowledgeReadiness, scoreOrigin, scorePrReviewComments, scorePrReviewSource, scoreRedTeamOutput, scoreReferenceReplay, scoreTraceInsightReadiness, seatPresets, securityJudge, selectHarnessVariant, selfPreference, sentenceReorderMutator, serializeFeedbackTrajectoriesJsonl, servedModelAcceptable, showMeasured, signManifest, spearmanR, statusAdvanced, stopOnNoProgress, stopOnRepeatedAction, stringField, stripFencedJson, studentTCdf, subjectiveEval, summarizeAgentReceiptIntegrity, summarizeBackendIntegrity, summarizeHarnessResults, summarizePrReviewBenchmark, summarizePreferenceMemory, summaryTable, supervisorRunRolloutLines, testJudge, textInSnapshot, throwIfRunIncomplete, toAgentProfileJson, toHarborTrajectories, toHarborTrajectory, toJsonl, toLangfuseEnvelope, toOpenAiTool, toPrometheusText, toRewardRows, toSftRows, tokenizeDomainWords, toolNamesForRun, toolSpans, toolSpansToTraceAnalysisStore, traceAnalystFunctionGroup, traceAnalystOnRunComplete, traceContract, traceJudge, traceJudgeEnsemble, traceSpanKindToOpenInferenceKind, tracedAnalyzeTraces, trainingReward, trainingScore, typoMutator, unmintableReasons, urlContains, userQuestionsForKnowledgeGaps, validateAgentProfileCell, validateProductBenchmarkManifest, validateProductBenchmarkRecord, validateProductBenchmarkRun, validateRolloutLine, validateRunRecord, verbosityBias, verifyAgentProfileCell, verifyAttestation, verifyCompletion, verifyManifest, visualDiff, viteDeployRunner, vitestTestParser, warnDeprecatedOnce, weightedComposite, weightedMean, weightedRecall, welchsTTest, whitespaceCollapseMutator, wilcoxonSignedRank, wilson, withAssignedFeedbackSplit, withHeldoutBlend, withJudgeRetry, withOtelPipeline, wranglerDeployRunner, writeSupervisorRunReport };
|
|
5747
|
+
export { AGENT_PROFILE_KINDS, ATIF_SCHEMA_VERSION, ATTESTATION_ALGORITHM, type ActionExecutionPolicy, type ActionPolicyDecision, type ActionableSideInfo, type ActiveLearningOptions, type AdapterRun, AgentDriver, type AgentDriverConfig, AgentEvalError, type AgentEvalErrorCode, type AgentInterfaceProfileLike, type AgentProfile, type AgentProfileCell, type AgentProfileCellInput, type AgentProfileCellSchemaVersion, AgentProfileCellValidationError, type AgentProfileDimensionValue, type AgentProfileHarness, type AgentProfileJson, type AgentProfileJsonObject, type AgentProfileKind, type AgentProfileRuntimeReceipt, type AgentProfileSource, type AgentProfileSourceInput, type AlignmentOp, type Analyst, type AnalystContext, type AnalystCost, type AnalystFeedbackTrajectoryOptions, type AnalystFinding, type AnalystFindingDigest, type AnalystHooks, type AnalystInputKind, type AnalystMissedIssue, AnalystRegistry, type AnalystRegistryOptions, type AnalystRequirements, type AnalystReviewCounts, type AnalystReviewDecision, type AnalystReviewQuality, type AnalystReviewRequest, type AnalystReviewSource, type AnalystRunDigest, type AnalystRunEvent, type AnalystRunInputs, type AnalystRunResult, type AnalystRunSummary, type AnalystSeverity, type AnalystUsageReceipt, type AnalyzeTracesInput, type AnalyzeTracesOptions, type AnalyzeTracesResult, type AntiSlopConfig, type AntiSlopIssue, type AntiSlopReport, Artifact, type ArtifactCheck, type Artifact$1 as ArtifactCheckArtifact, type ArtifactEventLike, type ArtifactResult, type ArtifactValidator, type AsiSeverity, type AssertCapabilityHeadroomOptions, type AssertCrossFamilyOptions, type AssertCrossFamilyServedOptions, type AssertModelsServedOptions, type AssertServedModelOptions, type AssertSingleBackendOptions, type AttestationProvenance, type AttestationVerification, type AttestedReport, type AutoPrClient, BENCHMARK_SPLIT_SEED, BOOTSTRAP_GATE_MIN_N, type BackendDescriptor, BackendIntegrityError, type BackendIntegrityReport, type BaselineOptions, type BaselineReport, type BehaviorAssertion, type BehavioralMetrics, type BehavioralTokenSequence, type BenchmarkAdapter, type BenchmarkDatasetItem, type BenchmarkEvaluation, type BenchmarkFamily, type BenchmarkReport, type BenchmarkResponder, BenchmarkRunner, type BenchmarkRunnerConfig, type BenchmarkScenario, type BenchmarkSource, type BenchmarkTaskKind, type BisectOptions, type BisectResult, type BisectStep, type BlendWeights, type BootstrapOptions, type BootstrapResult, type BoundedTraceAnalysisStoreOptions, BudgetBreachError, BudgetGuard, BudgetLedgerEntry, type BudgetPolicy, BudgetSpec, type BuildTraceAnalysisToolsOptions, CODING_HARNESSES, CONTROL_INTEGRITY_ANALYST, type CachedJudge, type CachedJudgeOptions, type CalibrationResult, type CallExpectation, CallbackResearcher, type CallbackResearcherOptions, type CampaignFactoryParams, type CampaignIntegrityPolicy, type CampaignRunContext, type CampaignRunOutcome, type CampaignRunner, type CampaignScenario, type CampaignVariant, type CanaryAlert, type CanaryEvaluation, type CanaryKind, type CanaryLeak, type CanaryOptions, type CanaryReport, type CanarySeverity, type CandidateComparison, type CandidateScenario, type CandidateScore, type CapabilityHeadroomOptions, type CapabilityHeadroomResult, CaptureFetchContext, CaptureFetchOptions, CaptureIntegrityError, type CausalAttributionReport, type CellVerdict, type ChannelRollup, type ChatCallOpts, type ChatClient, type ChatMessage, type ChatRequest, type ChatResponse, type ChatToolCall, type ChatTransport, type CheckResult, type CheckerIdentity, type CheckerOutcome, type CliBridgeTransportOpts, type CliffsMagnitude, type ClusterBootstrapInterval, type ClusterSignFlipAlternative, type ClusterSignFlipResult, type ClusteredBinaryCluster, type ClusteredMatchedPair, type ClusteredPairedBinaryOptions, type ClusteredPairedBinaryResult, type ClusteredPairedBinaryStatistics, type CollectedArtifacts, type CommandRunner, type ComparePairedArmsOptions, type CompletionCriterion, type CompletionRequirement, type CompletionVerdict, type ConceptComplexity, type ConceptFinding, type ConceptSpec, type ConceptWeightStrategy, ConfigError, type ContinuityCheck, type ContinuityCheckResult, type ContinuityReport, type ContinuitySnapshotPair, type ContinuousAgreement, type ContinuousAgreementOptions, type ContinuousCalibrationResult, type ContractCheckResult, type ContractJudgeOptions, type ContractMetric, type ContractReport, type ContractRule, type ContractRuleKind, type ContractSpan, type ContractVerdict, type ContractViolation, type ControlActionFailureMode, type ControlActionOutcome, type ControlBudget, type ControlContext, type ControlDecision, type ControlEvalResult, ControlIntegrityAnalyst, type ControlRunResult, type ControlRunToRunRecordOptions, type ControlRuntimeConfig, type ControlRuntimeError, type ControlSeverity, type ControlStep, type ControlStopPolicies, ConvergenceTracker, type CorpusAgreementOptions, type CorpusAgreementPerDimension, type CorpusAgreementReport, type CorpusScoreRecord, type CorrectnessChecker, CostAccountingIncompleteError, CostCallConflictError, CostCeilingReachedError, type CostChannel, type CostEntry, CostLedger, type CostLedgerFilter, type CostLedgerHandle, type CostLedgerOptions, type CostLedgerPersistence, CostLedgerPersistenceError, type CostLedgerSummary, type CostProvenance, type CostReceipt, CostReceiptCaptureError, type CostReceiptInput, type CostReport, CostReservationExceededError, type CostResult, type CostSummary, CostTracker, type CostUsage, type CounterfactualContext, type CounterfactualMutation, type CounterfactualResult, type CounterfactualRunner, type CreateChatClientOpts, type CreateDefaultReviewerOptions, type CreateExperimentInput, type CreateSandboxPoolOpts, type CreateTraceAnalystOptions, CrossFamilyError, type CrossTraceDiff, type CrossTraceDiffOptions, type CustomTokenPricing, type CustomTransportOpts, DECISION_PAIRED_DELTA_STATISTIC, DEFAULT_AGENT_SLOS, DEFAULT_COMPLEXITY_WEIGHTS, DEFAULT_RULES as DEFAULT_FAILURE_RULES, DEFAULT_FINDERS, DEFAULT_HARNESS_OBJECTIVES, DEFAULT_MUTATION_PRIMITIVES, DEFAULT_MUTATORS, DEFAULT_PERMUTATIONS, DEFAULT_PR_REVIEW_SCORE_WEIGHTS, DEFAULT_REDACTION_RULES, DEFAULT_RED_TEAM_CORPUS, DEFAULT_RUN_SCORE_WEIGHTS, DEFAULT_SEVERITY_WEIGHTS, DEFAULT_TRACE_ANALYST_BUDGETS, DEFAULT_TRACE_ANALYST_KINDS, DEFAULT_TRACE_ANALYST_LIMITS, DataAcquisitionPlan, Dataset, type DatasetDifficulty, type DatasetManifest, type DatasetOverview, type DatasetProvenance, type DatasetScenario, type DatasetSplit, type DecideNextUserTurnOpts, type DefaultAnalystRegistryOptions, type DefaultVerdict, type DeltaStatistic, type DeployFamily, type DeployGateLayerInput, type DeployRunResult, type DeployRunner, type DescriptionLengthCandidate, type DescriptionLengthConfig, type DescriptionLengthDecision, type DescriptionLengthEvidence, DescriptionLengthGate, type DescriptionLengthRejectionCode, type DetectorEvent, type DetectorSeverity, type DetectorSignal, type DiffPolicy, type DiffScorecardOptions, type DirEntry, type DirectProviderTransportOpts, type Direction, type DiscoverPersonasOptions, type DiscoveredPersona, DockerSandboxDriver, type DriverResult, type DriverState, type DspyRlmTraceEngineOptions, DualAgentBench, type DualAgentBenchConfig, type DualAgentReport, type DualAgentRound, type DualAgentScenario, type DualAgentScenarioResult, type EProcess, type EProcessOptions, type EProcessState, type EProcessStep, ERROR_COUNT_PATTERNS, type EnsembleAggregate, type EnsembleJudgeOptions, type EquivalenceArm, type EquivalenceCheckDefinition, type EquivalenceCheckSpec, type EquivalenceChecker, type EquivalenceCheckerInput, type EquivalenceCheckerResult, type EquivalenceObligation, type EquivalenceObligationStatus, EquivalenceProtocolError, type EquivalenceRecord, type ErrorCluster, type ErrorCountPattern, type ErrorStreakOptions, type EvalCampaignOptions, type EvalCampaignResult, type EvalResult, type EvalToolDef, EvalTraceStore, EventFilter, EventKind, type EvidenceRef, type EvolutionRound, type ExactAnalystBudgetPolicy, type ExactAnalystBudgetSnapshot, type ExactAnalystExecutionPlanSnapshot, type ExactAnalystRunCompletion, type ExactAnalystRunEvent, ExactAnalystRunExecutionError, type ExactAnalystRunPolicySnapshot, type ExactAnalystRunResult, type ExactAnalystRunSummary, type ExactAnalystSnapshot, type ExactCapableAnalyst, type ExactExecutionComponentIdentity, type ExactExecutionComponentSnapshot, type ExactRegistryRunOpts, type ExactRiskDifferenceResult, type ExecutionProbe, type ExecutionProbeOutcome, type ExecutionProbeRequest, type ExecutorConfig, type Expectation, type Experiment, type ExperimentPlan, type ExperimentProvenance, type ExperimentRep, type ExperimentResult, type ExperimentStats, type ExperimentStore, ExperimentTracker, type ExperimentTrackerOptions, type ExperimentVerdict, ExportableSpan, type ExportedRewardModel, type ExtractOptions, type ExtractResult, ExtractUsageFromSseOptions, ExtractedUsage, FAILURE_CLASSES, FAILURE_MODE_KIND_SPEC, type FactorContribution, type FactorialCell, type FailedRun, FailureClass, type FailureClassification, type FailureContext, type FailureMode, type FailureRule, type FeedbackArtifactType, type FeedbackAttempt, type FeedbackLabel, type FeedbackLabelKind, type FeedbackLabelSource, type FeedbackOptimizerRow, type FeedbackOutcome, type FeedbackPattern, type FeedbackReplayAdapter, type FeedbackReplayResult, type FeedbackSeverity, type FeedbackSplitPolicy, type FeedbackTask, type FeedbackTrajectory, type FeedbackTrajectoryFilter, type FeedbackTrajectoryStore, type FieldDestination, type FileChange, FileSystemFeedbackTrajectoryStore, FileSystemRawProviderSink, FileSystemRawProviderSinkOptions, FileSystemTraceStore, FileSystemTraceStoreOptions, type Finding, type FindingSubject, type FindingSubjectKind, type FindingsDiff, FindingsStore, type FlattenOtlpOptions, type FlowAction, type FlowLayerEnv, type FlowLayerFactoryInput, type FlowRunner, type FlowRunnerStepResult, type FlowSpec, type FlowStep, type FromHarborOptions, type GainDistributionBin, type GainDistributionFigureSpec, type GainDistributionOptions, type GateDecision, type GateEvidence, GenericSpan, type GhCliClientOptions, type GoldenItem, type GoldenSeverity, type GoldenSpec, HARBOR_IMPORT_GAP, HARNESS_BRIEFS, HARNESS_NATIVE_MODEL, type HarborAgent, type HarborContentPart, type HarborFinalMetrics, type HarborImageSource, type HarborMetrics, type HarborObservation, type HarborObservationResult, type HarborStep, type HarborStepSource, type HarborSubagentTrajectoryRef, type HarborToolCall, type HarborTrajectory, type HarnessAdapter, type HarnessConfig, type HarnessExperimentConfig, type HarnessExperimentResult, type HarnessIntervention, type HarnessRunRequest, type HarnessRunResult, type HarnessScenario, type HarnessSelection, type HarnessType, type HarnessVariant, type HarnessVariantReport, type HeadroomClass, type HeadroomInput, HeldOutGate, type HeldOutGateConfig, type HeldOutGateRejectionCode, type HeldOutPartition, type HiddenCriteriaGrader, type HiddenGradeResult, type HiddenLeak, HoldoutAuditor, HoldoutLockedError, type HttpGithubClientOptions, type HypothesisManifest, type HypothesisResult, IMPROVEMENT_KIND_SPEC, INPUT_VALUE, INTENT_MATCH_JUDGE_VERSION, type ImageData, type ImprovementThresholds, type ImprovementVerdictResult, InMemoryFeedbackTrajectoryStore, InMemoryRawProviderSink, InMemoryRawProviderSinkOptions, InMemoryTraceStore, InMemoryWorkspaceInspector, type InferenceScorer, type InspectorContext, type IntentMatchInput, type IntentMatchOptions, type IntentMatchResult, type InteractionContribution, type InterimReleaseConfidence, type InterimReleaseConfidenceInput, type JudgeConfig, JudgeError, type JudgeFamily, type JudgeFleetOptions, type JudgeFn, type JudgeInput, JudgeParseError, type JudgeReplayGateArgs, type JudgeReplayResult, type JudgeRetryOutcome, type JudgeRetryPolicy, type JudgeRubric, JudgeRunner, type JudgeScore, type JudgeScoreInput, type JudgeScoresRecord, JudgeSpan, type JudgeVerdict, KNOWLEDGE_GAP_KIND_SPEC, KNOWLEDGE_POISONING_KIND_SPEC, type KeywordConceptSpec, type KeywordCoverageFinding, type KeywordCoverageOptions, type KeywordCoverageResult, KnowledgeAcquisitionMode, KnowledgeBundle, KnowledgeFallbackPolicy, KnowledgeFreshness, KnowledgeImportance, KnowledgeReadinessReport, KnowledgeRecommendedAction, KnowledgeRequirement, KnowledgeRequirementCategory, KnowledgeResponsibleSurface, KnowledgeSensitivity, LLM_CACHED_TOKENS, LLM_CACHED_TOKEN_ATTR_KEYS, LLM_CACHE_WRITE_TOKENS, LLM_CACHE_WRITE_TOKEN_ATTR_KEYS, LLM_CONTEXT_TOKENS, LLM_COST_ATTR_KEYS, LLM_COST_USD, LLM_INPUT_TOKENS, LLM_INPUT_TOKEN_ATTR_KEYS, LLM_MODEL_ATTR_KEYS, LLM_MODEL_NAME, LLM_OUTPUT_TOKENS, LLM_OUTPUT_TOKEN_ATTR_KEYS, LLM_REASONING_TOKENS, LLM_REASONING_TOKEN_ATTR_KEYS, type LangfuseEnvelope, type LangfuseGeneration, type LangfuseScore, type Layer, type LayerResult, type LayerStatus, type LeaderboardOptions, type LeaderboardRow, LimitExceededError, type LiveProofArtifact, type LiveProofConfig, type LiveProofContext, type LiveProofResult, LlmCallError, type LlmCallMetadata, type LlmCallRequest, type LlmCallResult, LlmClient, type LlmClientOptions, type LlmCorrectnessCheckerOpts, type LlmJsonCall, type LlmJudgeDimension, type LlmJudgeOptions, type LlmMessage, LlmResponseError, type LlmReviewerConfig, LlmRouteAssertionError, type LlmRouteRequirements, LlmSpan, LlmSpanOtlpInput, type LlmUsage, LockedJsonlAppender, MANN_WHITNEY_EXACT_MAX_STATES, MANN_WHITNEY_EXACT_MAX_WORK, MODEL_PRICING, type MakeEvalToolsConfig, type MannWhitneyResult, type MatchResult, type MatchedPair, type MatchedRunRecordPair, type MatcherResult, type MaximumCharge, type McNemarResult, type Measured, type MeasurementPolicy, type MergeOptions, Message, type MetricSamples, type MetricVerdict, MetricsCollector, type MintRolloutOptions, type MintRolloutResult, type MintedRolloutLine, type MintedRolloutOutcome, type MockTransportOpts, type ModelCostRollup, type ModelPreflight, type ModelSeats, ModelSubstitutionError, ModelsUnreachableError, type MuffledFinder, type MuffledFinding, MultiLayerVerifier, type MultiToolchainLayerConfig, type Mutator, Mutex, type NoLeakOptions, type NoProgressOptions, NoopRawProviderSink, NoopResearcher, NotFoundError, OPENINFERENCE_SPAN_KIND, OTEL_AGENT_EVAL_SCOPE, OUTPUT_VALUE, type Objective, type Oracle, type OracleObservation, type OracleReport, type OracleResult, type OrthogonalityInput, type OrthogonalityResult, OtelExportConfig, OtelExporter, type OtelPipelineHandle, type OtelPipelineOptions, OtlpExport, OtlpFileTraceStore, type OtlpFileTraceStoreOptions, type OtlpFlatLine, OtlpResourceSpans, OtlpSpan, OtlpSpanRole, OtlpSpanRoleInput, type OtlpToRunRecordsOptions, type OtlpTraceRunRecord, PROBE_MAX_TOKENS, type PaidCallResult, type PairArmsOptions, type PairArmsResult, type PairRunRecordsResult, type PairedArmRow, type PairedArmsComparison, type PairedBootstrapOptions, type PairedBootstrapResult, type PairedCorrectness, type PairedDecisionMethod, type PairedDecisionShape, type PairedDecisionStatistic, type PairedDeltaTestOptions, type PairedDeltaTestResult, type PairedEvalueOptions, type PairedEvalueSequence, type PairedEvalueStep, type PairedMcNemarEvidence, type PairedMetricDelta, type PairedPromotionDecision, type PairedPromotionDecisionOptions, type PairedSignTestResult, type PairedTTestResult, PairwiseSteeringOptimizer, type ParaphraseRobustnessScenarioInput, type ParaphraseRobustnessScenarioResult, type ParetoFigureSpec, type ParetoPoint, type ParetoResult, type PartitionHeldOutOptions, type PendingCostCall, type PendingCostCallView, type PersistedFinding, type PersonaConfig, type PersonaRigor, type Playbook, type PlaybookEntry, type PoolSlot, type PositionalBiasResult, type PrReviewAuditCase, type PrReviewBenchmarkSummary, type PrReviewComment, type PrReviewMatchedFinding, type PrReviewOutcome, type PrReviewReferenceFinding, type PrReviewScore, type PrReviewScoreWeights, type PrReviewSeverity, type PrReviewSource, type PreferenceMemoryEntry, type PreflightModelsOptions, type PreflightOutcome, type ProducedProposal, type ProducedState, type ProductBenchmarkArm, type ProductBenchmarkArtifactPaths, type ProductBenchmarkBudgets, type ProductBenchmarkExportOptions, type ProductBenchmarkExportResult, type ProductBenchmarkManifest, type ProductBenchmarkProfileRef, type ProductBenchmarkRecord, type ProductBenchmarkRepoRef, type ProductBenchmarkRunInput, type ProductBenchmarkScenario, type ProductBenchmarkSingleRunExportOptions, type ProductBenchmarkSplit, type ProductBenchmarkSubstrateVersions, type ProductBenchmarkValidationReport, ProductClient, type ProductClientConfig, type ProfileAxisSpec, type ProjectRuntimeTrajectoryEvidenceOptions, type ProjectedOtlpSpan, type PromptHandle, PromptRegistry, type ProportionInterval, type ProposalEventLike, type ProposalFinding, type ProposalFindingOrigin, type ProposeAutomatedPullRequestInput, type ProposeAutomatedPullRequestResult, type ProposeFn, type ProposeInput, type ProposeOutput, type ProposeReviewConfig, type ProposeReviewControlAction, type ProposeReviewControlConfig, type ProposeReviewControlResult, type ProposeReviewControlState, type ProposeReviewReport, type ProposeReviewShot, type ProposedSideEffect, type ProvenanceReader, ProviderRedactor, type QueryTracesPage, REDACTION_VERSION, REFERENCE_EQUIVALENCE_INPUT_LIMITS, REFERENCE_EQUIVALENCE_JUDGE_VERSION, RESEARCH_REPORT_HARD_PAIR_FLOOR, ROLLOUT_SCHEMA, RUN_COST_ATTR_KEYS, type RankTestMethod, type RankTestMethodRequest, type RankTestOptions, type RawAnalystEvidence, type RawAnalystFinding, RawProviderDirection, RawProviderEvent, RawProviderSink, RawProviderSinkFilter, type RecordRunsOptions, type RedTeamCase, type RedTeamCategory, type RedTeamFinding, type RedTeamPayload, type RedTeamReport, RedactionReport, RedactionRule, type ReferenceEquivalenceJudgeInput, type ReferenceEquivalenceJudgeOptions, type ReferenceEquivalenceJudgeResult, type ReferenceEquivalenceScenario, type ReferenceMatchResult, type ReferenceReplayAdapter, type ReferenceReplayAdapterFn, type ReferenceReplayAdapterLike, type ReferenceReplayAggregate, type ReferenceReplayCandidate, type ReferenceReplayCase, type ReferenceReplayCaseRun, type ReferenceReplayExecutionScenario, type ReferenceReplayItem, type ReferenceReplayMatch, type ReferenceReplayMatchStrategy, type ReferenceReplayMatcher, type ReferenceReplayPromotionDecision, type ReferenceReplayPromotionPolicy, type ReferenceReplayRun, type ReferenceReplayRunContext, type ReferenceReplayRunOptions, type ReferenceReplayRunStore, type ReferenceReplayScenario, type ReferenceReplayScenarioScore, type ReferenceReplayScore, type ReferenceReplayScoreOptions, type ReferenceReplaySplit, type ReferenceReplaySplitComparison, type ReferenceReplaySteeringRowsOptions, type ReflectionContext, type ReflectionProposal, type RegistryRunOpts, type ReleaseConfidenceAxis, type ReleaseConfidenceAxisName, type ReleaseConfidenceInput, type ReleaseConfidenceIssue, type ReleaseConfidenceMetrics, type ReleaseConfidenceScorecard, type ReleaseConfidenceStatus, type ReleaseConfidenceThresholds, type ReleaseTraceEvidence, type RenderReleaseReportOptions, type RepeatedActionOptions, ReplayCache, type ReplayCacheEntry, ReplayCacheMissError, type ReplayCacheStats, ReplayError, type ReplayFetchOptions, type RepoRef, type RequirementCheck, type ResearchReport, type ResearchReportCandidate, type ResearchReportDecision, type ResearchReportMethodology, type ResearchReportOptions, type ResearchReportRecommendation, type Researcher, RetrievalSpan, type Review, type ReviewFn, type ReviewInput, type ReviewMemoryEntry, type ReviewMemoryStore, type ReviewerMemoryEntry, type ReviewerOutput, type ReviewerPromptInput, type ReviewerSoftFailDefaults, type ReviewerVerificationSummary, type RewardRow, type RiskDifferenceResult, type RobustnessResult, type RolloutCapture, type RolloutLine, type RolloutRole, type RolloutScrubber, type RolloutSplit, type RolloutStep, type RouteMap, type RoutedField, type RouterTransportOpts, type RubricDimension, Run, type RunCommandInput, type RunCommandResult, RunCompleteHook, RunCompleteHookContext, type RunCostProvenance, RunCritic, type RunCriticOptions, type RunEvidenceMetadata, RunFilter, RunIntegrityError, RunIntegrityExpectations, RunIntegrityIssue, RunIntegrityIssueCode, RunIntegrityReport, type RunJudgeMetadata, RunLayer, RunOutcome, type RunPaidCallInput, type RunRecord, type RunRecordBackend, type RunRecordFilter, RunRecordValidationError, type RunScore, type RunScoreWeights, type RunSplitTag, RunStatus, type RunTaskFailure, type RunTerminalOutcome, type RunTokenUsage, type RunTrace, type RuntimeEventLike, type RuntimeResolution, type RuntimeTrajectoryEvidenceProjection, type RuntimeTrajectoryEvidenceSummary, type RuntimeTrajectoryHookEvent, type RuntimeTrajectoryRecord, type RuntimeTrajectoryRunRecord, SEMANTIC_CONCEPT_JUDGE_VERSION, SKILL_USAGE_ANALYST, SPAN_KIND_ATTR_KEYS, SUPERVISOR_RUN_INTEGRITY_SCHEMA, SUPERVISOR_RUN_SCHEMA, type SandboxDriver, SandboxHarness, type SandboxHarnessResult, type SandboxJudgeKind, type SandboxJudgeResult, type SandboxJudgeSpec, type SandboxPool, type SandboxResult, type SandboxSdkTransportOpts, SandboxSpan, type SatisfiedBy, type ScanOptions, type Scenario, type ScenarioCost, type ScenarioFile, ScenarioRegistry, type ScenarioResult, ScoreKnowledgeReadinessOptions, type ScoreOrigin, type ScorePreference, type ScoreRiskDifferenceResult, type Scorecard, type ScorecardCell, type ScorecardCellDiff, type ScorecardDiff, type ScorecardEntry, type ScorecardLogLine, type ScoredTarget, type SearchSpanResult, type SearchTraceResult, type SeatName, type SeatPresetName, SeatUnsetError, type SelfPlayOptions, type SelfPlayProposer, type SelfPlayScorer, type SelfPreferenceResult, type SemanticConceptJudgeInput, type SemanticConceptJudgeOptions, type SemanticConceptJudgeResult, type SequentialDecision, type SerializedRegex, type SeriesConvergenceOptions, type SeriesConvergenceResult, ServedCrossFamilyError, type ServedModelCheck, type ServedModelVerdict, type Severity, type SftExportOptions, type SftRow, type SignTestAlternative, type SignedManifest, type SignedManifestAlgo, type SingleBackendDivergence, SingleBackendError, type SingleBackendField, type SingleBackendReport, SkillUsageAnalyst, type SliceOptions, type Slo, type SloCheckResult, type SloComparator, type SloReport, type SloSeverity, type SlopCategory, type SlotFactory, type SourceLimits, Span, SpanBase, SpanFilter, SpanHandle, SpanKind, type SpanMatchRecord, SpanNotFoundError, type SpanPredicate, SpanStatus, type SplitCoverage, SseUsageMode, type SteeringBundle, type SteeringChange, type SteeringDelta, type SteeringOptimizationResult, type SteeringOptimizationRow, type SteeringOptimizationSelector, type SteeringOptimizerBackend, type SteeringOptimizerConfig, type SteeringRolePrompt, type StepAttribution, type StopDecision, type StrategyChecker, type StreamingDetector, type SuboptimalCode, type SuboptimalSignal, SubprocessSandboxDriver, type SubprocessSandboxDriverOptions, type SummaryTable, type SummaryTableOptions, type SummaryTableRow, type SupervisorRunIntegrityEvidence, type SupervisorRunIntegrityIssue, type SupervisorRunIntegrityIssueCode, type SupervisorRunIntegrityOptions, type SupervisorRunIntegrityReport, type SupervisorRunIntegritySeverity, type SupervisorRunNodeRole, type SupervisorRunReader, type SupervisorRunReport, type SupervisorRunRollup, type SupervisorRunSourceOnlyCheckCode, type SupervisorRunSources, type SupervisorRunTree, type SupervisorRunTreeGap, type SupervisorRunTreeGapCode, type SynthesisReason, type SynthesisTarget, TOOL_ARGS_CAPTURED, TOOL_LATENCY_MS, TOOL_NAME, TOOL_NAME_ATTR_KEYS, TRACE_ANALYSIS_LIMITS, TRACE_ANALYST_ACTOR_DESCRIPTION, TRACE_ANALYST_ACTOR_DESCRIPTION_VERSION, TRACE_ANALYST_TOOL_NAMESPACE, TRACE_ANALYST_TRUNCATION_MARKER_PREFIX, TRACE_SCHEMA_VERSION, type TaskGold, type TaskHeadroom, type TestGradedRunOptions, type TestGradedRunResult, type TestGradedScenario, type TestOutputParser, type TestResult, type TextMatcher, type ThresholdContract, TokenCounter, type TokenSpec, type ToolCallEventLike, type ToolDef, type ToolMatcher, ToolSpan, ToolSpanOtlpInput, type ToolSpansToTraceAnalysisStoreOptions, type ToolStats, ToolTraceMissingError, type ToolUseMetrics, type ToolUseOptions, type TraceAggregate, type TraceAnalysisEngine, type TraceAnalysisEngineRequest, type TraceAnalysisEngineResult, TraceAnalysisLimitError, type TraceAnalysisStore, type TraceAnalysisStoreContext, TraceAnalysisStoreContractError, type TraceAnalysisToolDescriptor, TraceAnalysisValidationError, type TraceAnalystByteBudgets, type TraceAnalystDefinition, type TraceAnalystFilters, type TraceAnalystHookOptions, type TraceAnalystLimits, type TraceAnalystSpan, type TraceAnalystSpanKind, type TraceAnalystSpanStatus, type TraceAnalystTraceSummary, type TraceContract, TraceContractBuilder, TraceEmitter, TraceEmitterOptions, TraceEvent, TraceFileMalformedError, TraceFileMissingError, TraceFileTooLargeError, type TraceInsightContext, type TraceInsightFinding, type TraceInsightPanelRole, type TraceInsightPromptInput, type TraceInsightQualityGate, type TraceInsightQuestion, type TraceInsightReadiness, type TraceInsightSuite, type TraceInsightTask, TraceNotFoundError, TraceStore, TraceStoreSource, TraceStoreToOtlpOptions, type TracedAnalystOptions, type TracedJudgeOptions, TracesToOtlpResult, type Trajectory, type TrajectoryStep, type TreatmentClass, type TreatmentGate, type TreatmentGateInput, type TreatmentGateOptions, type TrialTrace, type Turn, type TurnMetrics, type TurnResult, UI_FINDING_SEVERITIES, UI_LENSES, UNIVERSAL_FINDERS, type UiFinding, type UiFindingScreenshot, type UiFindingSeverity, type UiLens, type Unavailable, UserQuestion, VERIFICATION_STRATEGIES, VERIFICATION_STRATEGY_SOURCES, type ValidationContext, ValidationError, type ValidationIssue, type ValidationResult, type VerbosityBiasResult, type Verdict, type VerdictCacheStats, type VerdictCacheStore, type VerdictCertification, type Verification, VerificationError, type VerificationReport, type VerificationStrategyProfile, type VerificationStrategySource, type VerifyContext, type VerifyFn, type VerifyOptions, type ViewSpansResult, type ViewTraceOversized, type ViewTraceResult, type VisualDiffOptions, type VisualDiffResult, type ViteDeployRunnerInput, WILCOXON_EXACT_MAX_N, WORKER_DRIVER_DOCTRINE, type WeightedCompositeInput, type WeightedCompositeResult, type WelchTestResult, type WelchTestStatus, type WilcoxonSignedRankResult, type WorkflowTopology, type WorkspaceAssertion, type WorkspaceAssertionResult, type WorkspaceInspector, type WorkspaceSnapshot, type WranglerDeployRunnerInput, acquisitionPlansForKnowledgeGaps, agentProfileCellHashMaterial, agentProfileCellKey, agentProfileHash, agentProfileId, agentProfileModelId, agentVisibleFields, aggregateJudgeVerdicts, aggregateLlm, aggregatePrReviewScore, aggregateRunScore, allCriticalPassed, analystFindingDigest, analystRunDigest, analystRunToFeedbackTrajectory, analystRunToReviewRequests, analyzeAntiSlop, analyzeSeries, analyzeSupervisorRun, analyzeSupervisorRunIntegrity, analyzeSupervisorRunSources, analyzeTraces, appendScorecard, applyLlmSpanOtlpAttributes, applyToolSpanOtlpAttributes, argHash, asNumber, asString, assertCapabilityHeadroom, assertCrossFamily, assertCrossFamilyServed, assertExactRegistryRunOpts, assertLlmRoute, assertMinted, assertMintedLines, assertModelsServed, assertNoHiddenLeak, assertProductBenchmarkRun, assertRealAgentReceipts, assertRealBackend, assertReleaseConfidence, assertRolloutLine, assertRunAgentProfileCell, assertRunCaptured, assertServedModel, assertServedModels, assertSingleBackend, assignFeedbackSplit, assignHeldOutTag, attachCostToReport, attest, attributeCounterfactuals, backoffMs, deterministicSplit as benchmarkDeterministicSplit, index_d_exports as benchmarks, benjaminiHochberg, bisect, blendHeldout, blockingKnowledgeEval, bonferroni, bootstrapCi, buildAgentInterfaceProfileCell, buildAgentProfileCell, buildDefaultAnalystRegistry, buildDriverSystemPrompt, buildEquivalenceRecord, buildProductBenchmarkManifest, buildReflectionPrompt, buildReviewerPrompt, buildTraceAnalysisToolDescriptors, buildTraceInsightContext, buildTraceInsightPrompt, buildTrajectory, byteLengthRange, cachedJudge, calibrateJudge, calibrateJudgeContinuous, callLlm, callLlmJson, canaryLeakView, canonicalJson, canonicalize, capabilityHeadroom, captureFetchToRawSink, causalAttribution, checkBehavioralCanary, checkCanaries, checkServedModel, checkSlos, checkTraceContracts, clamp01, classifyFailure, classifyOtlpSpanRole, classifyTreatment, claudeCodeSupervisorRunReader, cliffsDelta, clusteredPairedBinary, cohensD, collectionPreserved, commentsForSource, commitBisect, comparePairedArms, compareReferenceReplay, compareToBaseline, compilerJudge, completionVerdict, composeParsers, composeValidators, computeExperimentStats, computeFindingId, computeToolUseMetrics, computeTraceMetrics, confidenceInterval, containsAll, contentHash, contextInputTokens, continuousAgreement, contractJudge, controlFailureClassFromVerification, controlRunToFeedbackTrajectory, controlRunToRunRecord, convertTraceStoresToOtlp, corpusInterRaterAgreement, corpusInterRaterAgreementFromJudgeScores, costForTokenPricing, costForUsage, costReceiptFromLlm, costReceiptFromLlmError, costReport, createAntiSlopJudge, createBoundedTraceAnalysisStore, createChatClient, createDefaultReviewer, createDspyRlmTraceEngine, createFeedbackTrajectory, createIntentMatchJudge, createLlmCorrectnessChecker, createLlmReviewer, createOtelExporter, createOtelTracingStore, createReferenceEquivalenceJudge, createReplayFetch, createSandboxPool, createSemanticConceptJudge, createTokenRecallChecker, createTraceAnalyst, crossTraceDiff, crowdingDistance, dataDescriptionBits, decideNextUserTurn, decidePairedPromotion, decideReferenceReplayPromotion, decideReferenceReplayRunPromotion, defaultBlendWeights, defaultIsMaterial, defaultProviderRedactor, defaultReferenceReplayMatcher, defaultTraceInsightPanel, defineEquivalenceCheck, defineTraceAnalyst, deployGateLayer, describeTraceInsightScope, diffFindings, diffScorecard, discoverPersonas, distillPlaybook, domainEvidencePattern, dominates, eProcess, emitControlIntegrityFindings, ensembleJudge, errorStreakDetector, estimateCost, estimateTokens, evaluateActionPolicy, evaluateContract, evaluateHypothesis, evaluateInterimReleaseConfidence, evaluateOracles, evaluateReleaseConfidence, evaluateTraceContract, executeScenario, expandProfileAxes, expectAgent, exportProductBenchmark, exportProductBenchmarkRuns, exportRewardModel, exportRunAsOtlp, extractAssetUrls, extractErrorCount, extractOtlpAttributes, extractProducedState, extractUsage, extractUsageFromResponse, extractUsageFromSse, feedbackTrajectoriesToDatasetScenarios, feedbackTrajectoriesToOptimizerRows, feedbackTrajectoryToDatasetScenario, feedbackTrajectoryToOptimizerRow, fileContains, fileExists, fileExperimentStore, fileVerdictCache, findAutoMatchNoExpectation, findConstructorCwdDropped, findFallbackToPass, findLiteralTruePass, findProductBenchmarkArtifacts, findSkipCountsAsPass, firstNumberAttr, firstStringAttr, flattenOtlpExportToNdjson, flowLayer, fnv1a32, formatBenchmarkReport, formatDriverReport, formatFindings, formatScorecardDiff, fromHarborTrajectory, gainHistogram, gateTreatmentApplied, gateTreatmentFromMetrics, gateTreatmentFromSpans, gateTreatmentFromToolSpans, ghCliClient, gitProvenanceReader, precision as goldenPrecision, gradeOnHidden, gradeSemanticStatus, groupBy, groupRunsByAgentProfileCell, harnessAxisOf, hasCapturedToolArgs, hashContent, hashJson, hashScenarios, hashToUnit, hiddenGrade, holm, htmlContainsElement, httpGithubClient, improvementVerdict, inMemoryExperimentStore, inMemoryReferenceReplayStore, inMemoryReviewStore, inMemoryRunRecordBackend, inMemoryVerdictCache, inferDomainKeywords, inferOtlpKind, interRaterReliability, interpretCliffs, iqr, isBinaryOutcomeVector, isHiddenDestination, isJudgeSpan, isLlmSpan, isModelPriced, isOtelConfigured, isOtlpModelCall, isRealnessGated, isRetrievalSpan, isRolloutLine, isRunRecord, isRuntimeSupervisorRunDir, isSandboxSpan, isToolSpan, isTrainableSplit, isTransientLlmError, isUnavailable, iterateRawCalls, jestTestParser, jsonHasKeys, jsonShape, jsonlReferenceReplayStore, jsonlReviewStore, jsonlRunRecordBackend, judgeFamily, judgeReplayGate, judgeSpans, keyPreserved, knowledgeReadinessTracePayload, leaderboard, linterJudge, llmJudge, llmSpanFromProvider, llmSpans, loadScorecard, loadScorerFromGrader, localCommandRunner, lowercaseMutator, makeEvalTools, makeFinding, makeProposalFinding, mannWhitneyU, mapConcurrent, matchGoldens, matchSpan, maximumChargeForLlmRequest, mcnemar, mcnemarPower, mcnemarRequiredN, mergeLayerResults, mergeSteeringBundle, minimumPairsForPairedDeltaTest, mintRolloutRows, modelDescriptionBits, modelHasSnapshot, modelPriceKey, mulberry32, multiToolchainLayer, noProgressDetector, normalCdf, normalizeModelId, normalizeScores, notBlocked, objectiveEval, observeAll, observedScore, observedSplitScore, otelRunCompleteHook, otlpRowsToRunRecords, otlpRowsToTraceRunRecords, otlpTextToTraceAnalysisStore, otlpToRunRecords, otlpToTraceRunRecords, pairArms, pairRunRecords, pairedBinaryScale, pairedBootstrap, pairedCohensDz, pairedDecisionShape, pairedDeltaTest, pairedDeltaTieFraction, pairedEvalueSequence, pairedMde, pairedRiskDifference, pairedRiskDifferenceExact, pairedRiskDifferenceScore, pairedSignTest, pairedTTest, paraphraseRobustness, paraphraseRobustnessScenarios, paretoChart, paretoFrontier, paretoFrontierWithCrowding, parseCorrectnessResponse, parseFeedbackTrajectoriesJsonl, parseReflectionResponse, parseRunRecordSafe, parseRuntimeTrajectoryHookEvent, partialCredit, partitionHeldOut, passAtK, passOrthogonality, pearsonR, pixelDeltaRatio, planTraceInsightQuestions, politenessPrefixMutator, positionalBias, preflightModels, printDriverSummary, probeLlm, productBenchmarkIntegrityFailures, productBenchmarkMutableSurfaces, productBenchmarkRepoIdentity, productBenchmarkSplits, index_d_exports$1 as profile, projectOtlpFlatLine, projectRuntimeTrajectoryEvidence, promptBisect, proposeSynthesisTargets, providerFromBaseUrl, pytestTestParser, ranks, readClaudeCodeSupervisorRun, readOtlpStatus, readProductBenchmarkManifest, readProductBenchmarkRecords, readRuntimeSupervisorRun, recordRuns, recordRunsToScorecard, redTeamDataset, redTeamReport, redactString, redactValue, referenceReplayRunsToSteeringRows, referenceReplayScenarioToRunScore, regexMatch, regexMatches, relabelImportedSplit, renderMarkdownReport, renderPlaybookMarkdown, renderPreferenceMemoryMarkdown, renderPriorFindings, renderReleaseReport, renderSteeringText, renderSupervisorRunHeadline, renderSupervisorRunMarkdown, renderUpstreamFindings, repeatedActionDetector, replayFeedbackTrajectories, replayFeedbackTrajectory, replayScorerOverCorpus, replayTraceThroughJudge, requireAgentProfileCell, requiredPairedSampleSize, requiredSampleSize, researchReport, resetDeprecationWarnings, resolveModelPricing, resolveSeat, resolveTraceAnalystLimits, rollupSupervisorRuns, roundTripRunRecord, routeFields, rowCount, rowWhere, runAgentControlLoop, runAssertions, runBehavioralCanaries, runCanaries, runCounterfactual, runE2EWorkflow, runEquivalenceCheck, runEvalCampaign, runExpectations, runFailureClass, runHarnessExperiment, runIntentMatchJudge, runJudgeFleet, runKeywordCoverageJudge, runKeywordCoverageJudgeUrl, runLiveProof, runProposeReview, runProposeReviewAsControlLoop, runRecordToProductBenchmarkRecord, runReferenceEquivalenceJudge, runReferenceReplay, runScore, runSelfPlay, runSemanticConceptJudge, runTaskScore, runTestGradedScenario, runTraceAnalyst, runsForScenario, runtimeSupervisorRunReader, scalarScore, scanForMuffledGates, scoreContinuity, scoreFromEvals, scoreKnowledgeReadiness, scoreOrigin, scorePrReviewComments, scorePrReviewSource, scoreRedTeamOutput, scoreReferenceReplay, scoreTraceInsightReadiness, seatPresets, securityJudge, selectHarnessVariant, selfPreference, sentenceReorderMutator, serializeFeedbackTrajectoriesJsonl, servedModelAcceptable, showMeasured, signManifest, spearmanR, statusAdvanced, stopOnNoProgress, stopOnRepeatedAction, stringField, stripFencedJson, studentTCdf, subjectiveEval, summarizeAgentReceiptIntegrity, summarizeBackendIntegrity, summarizeHarnessResults, summarizePrReviewBenchmark, summarizePreferenceMemory, summaryTable, supervisorRunRolloutLines, testJudge, textInSnapshot, throwIfRunIncomplete, toAgentProfileJson, toHarborTrajectories, toHarborTrajectory, toJsonl, toLangfuseEnvelope, toOpenAiTool, toPrometheusText, toRewardRows, toSftRows, tokenizeDomainWords, toolNamesForRun, toolSpans, toolSpansToTraceAnalysisStore, traceAnalystFunctionGroup, traceAnalystOnRunComplete, traceContract, traceJudge, traceJudgeEnsemble, traceSpanKindToOpenInferenceKind, tracedAnalyzeTraces, trainingReward, trainingScore, typoMutator, unmintableReasons, urlContains, userQuestionsForKnowledgeGaps, validateAgentProfileCell, validateProductBenchmarkManifest, validateProductBenchmarkRecord, validateProductBenchmarkRun, validateRolloutLine, validateRunRecord, verbosityBias, verifyAgentProfileCell, verifyAttestation, verifyCompletion, verifyManifest, visualDiff, viteDeployRunner, vitestTestParser, warnDeprecatedOnce, weightedComposite, weightedMean, weightedRecall, welchsTTest, whitespaceCollapseMutator, wilcoxonSignedRank, wilson, withAssignedFeedbackSplit, withHeldoutBlend, withJudgeRetry, withOtelPipeline, wranglerDeployRunner, writeSupervisorRunReport };
|
|
5938
5748
|
//# sourceMappingURL=index.d.ts.map
|