@tangle-network/agent-eval 0.108.1 → 0.110.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/dist/analyst/index.d.ts +10 -12
- package/dist/analyst/index.js +8 -11
- package/dist/analyst/index.js.map +1 -1
- package/dist/{analyze-runs-DJYpep3L.d.ts → analyze-runs-Dmz6LA9e.d.ts} +4 -4
- package/dist/{baseline-Bbid3WoO.d.ts → baseline-DsNteOgR.d.ts} +32 -2
- package/dist/belief-state/index.d.ts +6 -6
- package/dist/benchmarks/index.d.ts +4 -4
- package/dist/benchmarks/index.js +7 -8
- package/dist/builder-eval/index.d.ts +4 -4
- package/dist/builder-eval/index.js +1 -2
- package/dist/builder-eval/index.js.map +1 -1
- package/dist/{calibration-BPmzuVPk.d.ts → calibration-Dz8TQV4y.d.ts} +2 -2
- package/dist/campaign/index.d.ts +161 -20
- package/dist/campaign/index.js +15 -8
- package/dist/{chunk-LOJ2QVCE.js → chunk-2IY4ILP4.js} +2 -2
- package/dist/{chunk-LIEJUH2I.js → chunk-6PL5MGDL.js} +9 -9
- package/dist/{chunk-2OGPXHOB.js → chunk-7NX6ZSBG.js} +36 -7
- package/dist/chunk-7NX6ZSBG.js.map +1 -0
- package/dist/{chunk-OVPVM4JC.js → chunk-GTERJI6Q.js} +4 -4
- package/dist/{chunk-YEHAEDUD.js → chunk-IMWDSFUM.js} +604 -2
- package/dist/chunk-IMWDSFUM.js.map +1 -0
- package/dist/{chunk-JZXGWLK5.js → chunk-MHNQWM4I.js} +62 -6
- package/dist/chunk-MHNQWM4I.js.map +1 -0
- package/dist/{chunk-QRVS7MX4.js → chunk-OW47B5WA.js} +3 -5
- package/dist/{chunk-QRVS7MX4.js.map → chunk-OW47B5WA.js.map} +1 -1
- package/dist/{chunk-DBDRR6GF.js → chunk-PLOMR3HP.js} +48 -2
- package/dist/chunk-PLOMR3HP.js.map +1 -0
- package/dist/{chunk-GDZAWO2I.js → chunk-QFGTU7MT.js} +2 -2
- package/dist/{chunk-6SKVFBTR.js → chunk-RNB2NICW.js} +115 -13
- package/dist/chunk-RNB2NICW.js.map +1 -0
- package/dist/{chunk-V7HNA47Z.js → chunk-RSVSSZKF.js} +5 -5
- package/dist/{chunk-5PK3626Q.js → chunk-XRGOKCMO.js} +88 -17
- package/dist/chunk-XRGOKCMO.js.map +1 -0
- package/dist/{code-agent-session-rnJKlqmT.d.ts → code-agent-session-yitf9I-F.d.ts} +1 -1
- package/dist/contract/index.d.ts +20 -23
- package/dist/contract/index.js +11 -13
- package/dist/contract/index.js.map +1 -1
- package/dist/{control-B8UthSBL.d.ts → control-U8LBKUES.d.ts} +5 -6
- package/dist/control.d.ts +8 -9
- package/dist/control.js +6 -8
- package/dist/{dataset-DS7ytHZU.d.ts → dataset-NENEzRgk.d.ts} +1 -1
- package/dist/{default-registry-BswHCXnU.d.ts → default-registry-Bcf1uKVI.d.ts} +1 -2
- package/dist/{emitter-C2rqGH_l.d.ts → emitter-BRchAAAx.d.ts} +2 -2
- package/dist/{failure-cluster-DH9Flgcf.d.ts → failure-cluster-C48PiReX.d.ts} +2 -2
- package/dist/feedback-trajectory-pDcz1lQ1.d.ts +348 -0
- package/dist/{gepa-B3x5Ulcv.d.ts → gepa-BUNP3606.d.ts} +143 -2
- package/dist/hosted/index.d.ts +7 -7
- package/dist/{index-pPtfoIJO.d.ts → index-Dc3VLGhp.d.ts} +2 -2
- package/dist/index.d.ts +645 -61
- package/dist/index.js +1282 -190
- package/dist/index.js.map +1 -1
- package/dist/{insight-report-B4xrdwEK.d.ts → insight-report-D4cXFsLt.d.ts} +1 -1
- package/dist/{integrity-DqGZg3st.d.ts → integrity-qemeBAyx.d.ts} +1 -1
- package/dist/{types-D1ytG0Yg.d.ts → kind-factory-20hcaYpf.d.ts} +169 -2
- package/dist/meta-eval/index.d.ts +5 -5
- package/dist/meta-eval/index.js +1 -2
- package/dist/meta-eval/index.js.map +1 -1
- package/dist/{multi-layer-verifier-CI4jdX-q.d.ts → multi-layer-verifier-BsqKuLyN.d.ts} +1 -1
- package/dist/multishot/index.d.ts +3 -3
- package/dist/openapi.json +1 -1
- package/dist/pipelines/index.d.ts +6 -7
- package/dist/pipelines/index.js +3 -6
- package/dist/pipelines/index.js.map +1 -1
- package/dist/{policy-edit-DQUXYMDm.d.ts → policy-edit-D2bBDZDf.d.ts} +2 -2
- package/dist/{pre-registration-BUhVPzE7.d.ts → pre-registration-BepVVa6P.d.ts} +3 -3
- package/dist/{provenance-DdDhf6cg.d.ts → provenance-DMvsfknv.d.ts} +3 -5
- package/dist/{query-0aTmbmQe.d.ts → query-Ck190MOd.d.ts} +2 -2
- package/dist/{release-report-DeJpsBiA.d.ts → release-report-oBfOz8ku.d.ts} +3 -3
- package/dist/reporting.d.ts +8 -8
- package/dist/{researcher-Wc7dx6GM.d.ts → researcher-CaH0CwFC.d.ts} +6 -6
- package/dist/rl.d.ts +568 -15
- package/dist/rl.js +4 -4
- package/dist/{rubric-predictive-validity-DPnyG-CE.d.ts → rubric-predictive-validity-C-fMteAW.d.ts} +1 -1
- package/dist/{run-record-I-Z3JNvO.d.ts → run-record-DksGsfgv.d.ts} +1 -1
- package/dist/{runtime-trajectory-iW9IhV3e.d.ts → runtime-trajectory-h5i0SZUj.d.ts} +1 -1
- package/dist/{schema-m0gsnbt3.d.ts → schema-SGWcK9wa.d.ts} +1 -1
- package/dist/{semantic-concept-judge-BmNZPB_j.d.ts → semantic-concept-judge-D7z6JCLZ.d.ts} +57 -4
- package/dist/{store-BcFXE6LG.d.ts → store-BsVi7ncX.d.ts} +1 -1
- package/dist/storyboard/index.d.ts +1 -1
- package/dist/{summary-report-QMZVe3P-.d.ts → summary-report-Bz-0-t8v.d.ts} +2 -2
- package/dist/{test-graded-scenario-DeODGLra.d.ts → test-graded-scenario-mzYBKspu.d.ts} +3 -3
- package/dist/traces.d.ts +54 -11
- package/dist/traces.js +25 -27
- package/dist/{types-BdIv5dvA.d.ts → types-v--ctu-b.d.ts} +2 -2
- package/dist/wire/index.d.ts +5 -6
- package/package.json +1 -71
- package/dist/adapters/http.d.ts +0 -142
- package/dist/adapters/http.js +0 -203
- package/dist/adapters/http.js.map +0 -1
- package/dist/adapters/langchain.d.ts +0 -95
- package/dist/adapters/langchain.js +0 -34
- package/dist/adapters/langchain.js.map +0 -1
- package/dist/adapters/otel.d.ts +0 -112
- package/dist/adapters/otel.js +0 -110
- package/dist/adapters/otel.js.map +0 -1
- package/dist/chunk-2OGPXHOB.js.map +0 -1
- package/dist/chunk-45EEMHTC.js +0 -35
- package/dist/chunk-45EEMHTC.js.map +0 -1
- package/dist/chunk-5BKGXME7.js +0 -65
- package/dist/chunk-5BKGXME7.js.map +0 -1
- package/dist/chunk-5PK3626Q.js.map +0 -1
- package/dist/chunk-6SK5VFYK.js +0 -100
- package/dist/chunk-6SK5VFYK.js.map +0 -1
- package/dist/chunk-6SKVFBTR.js.map +0 -1
- package/dist/chunk-DBDRR6GF.js.map +0 -1
- package/dist/chunk-DJWX3GVS.js +0 -81
- package/dist/chunk-DJWX3GVS.js.map +0 -1
- package/dist/chunk-FOUG2VVS.js +0 -855
- package/dist/chunk-FOUG2VVS.js.map +0 -1
- package/dist/chunk-JZXGWLK5.js.map +0 -1
- package/dist/chunk-K7QEIHHJ.js +0 -613
- package/dist/chunk-K7QEIHHJ.js.map +0 -1
- package/dist/chunk-KKHDIONI.js +0 -414
- package/dist/chunk-KKHDIONI.js.map +0 -1
- package/dist/chunk-KMPRBJK4.js +0 -74
- package/dist/chunk-KMPRBJK4.js.map +0 -1
- package/dist/chunk-Q2JRAWRI.js +0 -196
- package/dist/chunk-Q2JRAWRI.js.map +0 -1
- package/dist/chunk-RZTMDUO7.js +0 -49
- package/dist/chunk-RZTMDUO7.js.map +0 -1
- package/dist/chunk-STGVSCDH.js +0 -202
- package/dist/chunk-STGVSCDH.js.map +0 -1
- package/dist/chunk-YEHAEDUD.js.map +0 -1
- package/dist/control-runtime-Acf9CGhw.d.ts +0 -182
- package/dist/corpus-eBVwhCp1.d.ts +0 -560
- package/dist/counterfactual-DlOz8PBx.d.ts +0 -85
- package/dist/diagnose.d.ts +0 -252
- package/dist/diagnose.js +0 -382
- package/dist/diagnose.js.map +0 -1
- package/dist/feedback-trajectory-C9KCo8ag.d.ts +0 -169
- package/dist/governance/index.d.ts +0 -135
- package/dist/governance/index.js +0 -18
- package/dist/governance/index.js.map +0 -1
- package/dist/groundedness/index.d.ts +0 -112
- package/dist/groundedness/index.js +0 -77
- package/dist/groundedness/index.js.map +0 -1
- package/dist/harness-optimizer-mOl9XX_O.d.ts +0 -106
- package/dist/kind-factory-DvIGo_cP.d.ts +0 -171
- package/dist/knowledge/index.d.ts +0 -103
- package/dist/knowledge/index.js +0 -18
- package/dist/knowledge/index.js.map +0 -1
- package/dist/pareto-E-pembql.d.ts +0 -81
- package/dist/perf/index.d.ts +0 -123
- package/dist/perf/index.js +0 -18
- package/dist/perf/index.js.map +0 -1
- package/dist/prm/index.d.ts +0 -104
- package/dist/prm/index.js +0 -265
- package/dist/prm/index.js.map +0 -1
- package/dist/product-benchmark/index.d.ts +0 -247
- package/dist/product-benchmark/index.js +0 -37
- package/dist/product-benchmark/index.js.map +0 -1
- package/dist/red-team-KmmiqBlY.d.ts +0 -63
- package/dist/redact-B40YG2M_.d.ts +0 -45
- package/dist/rubric-Cc6UHvUb.d.ts +0 -73
- package/dist/run-critic-CmMf05uV.d.ts +0 -56
- package/dist/sink-fetch-B1Yg4Til.d.ts +0 -101
- package/dist/telemetry/file.d.ts +0 -19
- package/dist/telemetry/file.js +0 -45
- package/dist/telemetry/file.js.map +0 -1
- package/dist/telemetry/index.d.ts +0 -38
- package/dist/telemetry/index.js +0 -130
- package/dist/telemetry/index.js.map +0 -1
- package/dist/testing-C21CHsq2.d.ts +0 -20
- package/dist/testing.d.ts +0 -1
- package/dist/testing.js +0 -8
- package/dist/testing.js.map +0 -1
- package/dist/trajectory-2TkpSEVh.d.ts +0 -33
- package/dist/workflow/index.d.ts +0 -496
- package/dist/workflow/index.js +0 -2178
- package/dist/workflow/index.js.map +0 -1
- /package/dist/{chunk-LOJ2QVCE.js.map → chunk-2IY4ILP4.js.map} +0 -0
- /package/dist/{chunk-LIEJUH2I.js.map → chunk-6PL5MGDL.js.map} +0 -0
- /package/dist/{chunk-OVPVM4JC.js.map → chunk-GTERJI6Q.js.map} +0 -0
- /package/dist/{chunk-GDZAWO2I.js.map → chunk-QFGTU7MT.js.map} +0 -0
- /package/dist/{chunk-V7HNA47Z.js.map → chunk-RSVSSZKF.js.map} +0 -0
package/dist/index.d.ts
CHANGED
|
@@ -1,88 +1,70 @@
|
|
|
1
|
-
export { A as ActionExecutionPolicy, a as ActionPolicyDecision, C as ControlRunToRunRecordOptions, L as LlmJsonCall, b as LlmReviewerConfig, P as ProposeFn, c as ProposeInput, d as ProposeOutput, e as ProposeReviewConfig, f as ProposeReviewControlAction, g as ProposeReviewControlConfig, h as ProposeReviewControlResult, i as ProposeReviewControlState, j as ProposeReviewReport, k as ProposeReviewShot, R as Review, l as ReviewFn, m as ReviewInput, n as ReviewMemoryEntry, o as ReviewMemoryStore, p as RunEvidenceMetadata, V as Verification, q as VerifyFn, r as controlFailureClassFromVerification, s as controlRunToRunRecord, t as createLlmReviewer, u as evaluateActionPolicy, v as inMemoryReviewStore, w as jsonlReviewStore, x as runProposeReview, y as runProposeReviewAsControlLoop, z as scoreFromEvals } from './control-
|
|
2
|
-
import { R as RunRecord, b as RunSplitTag } from './run-record-
|
|
3
|
-
export { f as AGENT_PROFILE_KINDS, g as AgentInterfaceProfileLike, A as AgentProfileCell, e as AgentProfileCellInput, h as AgentProfileCellSchemaVersion, i as AgentProfileCellValidationError, j as AgentProfileDimensionValue, k as AgentProfileHarness, a as AgentProfileJson, l as AgentProfileKind, m as AgentProfileSource, n as AgentProfileSourceInput, J as JudgeScoresRecord, d as RunJudgeMetadata, o as RunOutcome, p as RunRecordValidationError, c as RunTokenUsage, q as agentProfileCellHashMaterial, r as agentProfileCellKey, s as assertRunAgentProfileCell, t as buildAgentInterfaceProfileCell, u as buildAgentProfileCell, v as groupRunsByAgentProfileCell, w as isRunRecord, x as modelHasSnapshot, y as parseRunRecordSafe, z as requireAgentProfileCell, B as roundTripRunRecord, C as toAgentProfileJson, D as validateAgentProfileCell, E as validateRunRecord, F as verifyAgentProfileCell } from './run-record-
|
|
4
|
-
import { B as BehavioralMetrics } from './semantic-concept-judge-
|
|
5
|
-
export {
|
|
6
|
-
import {
|
|
7
|
-
export { a as Analyst, b as AnalystContext,
|
|
8
|
-
export { a as AnalystHooks, A as AnalystRegistry, b as AnalystRegistryOptions, B as BudgetPolicy, D as DefaultAnalystRegistryOptions, R as RegistryRunOpts, c as buildDefaultAnalystRegistry } from './default-registry-
|
|
9
|
-
export {
|
|
10
|
-
export { F as FindingToPolicyEditOptions, P as POLICY_EDIT_AXES, a as POLICY_EDIT_TARGET_SURFACES, b as PolicyEdit, c as PolicyEditAdmission, d as PolicyEditAdmissionOptions, e as PolicyEditAxis, f as PolicyEditChange, g as PolicyEditExpectedGain, h as PolicyEditGainDirection, i as PolicyEditGainUnit, j as PolicyEditInit, k as PolicyEditRisk, l as PolicyEditSchemaVersion, m as PolicyEditSource, n as PolicyEditTarget, o as PolicyEditTargetSurface, p as PolicyEditValidationError, q as admitPolicyEdit, r as applyPolicyEditToSurface, s as computePolicyEditId, t as isPolicyEdit, u as makePolicyEdit, v as policyEditFromFinding, w as policyEditsFromFindings, x as scorePolicyEditReadiness, y as validatePolicyEdit } from './policy-edit-DQUXYMDm.js';
|
|
1
|
+
export { A as ActionExecutionPolicy, a as ActionPolicyDecision, C as ControlRunToRunRecordOptions, L as LlmJsonCall, b as LlmReviewerConfig, P as ProposeFn, c as ProposeInput, d as ProposeOutput, e as ProposeReviewConfig, f as ProposeReviewControlAction, g as ProposeReviewControlConfig, h as ProposeReviewControlResult, i as ProposeReviewControlState, j as ProposeReviewReport, k as ProposeReviewShot, R as Review, l as ReviewFn, m as ReviewInput, n as ReviewMemoryEntry, o as ReviewMemoryStore, p as RunEvidenceMetadata, V as Verification, q as VerifyFn, r as controlFailureClassFromVerification, s as controlRunToRunRecord, t as createLlmReviewer, u as evaluateActionPolicy, v as inMemoryReviewStore, w as jsonlReviewStore, x as runProposeReview, y as runProposeReviewAsControlLoop, z as scoreFromEvals } from './control-U8LBKUES.js';
|
|
2
|
+
import { R as RunRecord, b as RunSplitTag } from './run-record-DksGsfgv.js';
|
|
3
|
+
export { f as AGENT_PROFILE_KINDS, g as AgentInterfaceProfileLike, A as AgentProfileCell, e as AgentProfileCellInput, h as AgentProfileCellSchemaVersion, i as AgentProfileCellValidationError, j as AgentProfileDimensionValue, k as AgentProfileHarness, a as AgentProfileJson, l as AgentProfileKind, m as AgentProfileSource, n as AgentProfileSourceInput, J as JudgeScoresRecord, d as RunJudgeMetadata, o as RunOutcome, p as RunRecordValidationError, c as RunTokenUsage, q as agentProfileCellHashMaterial, r as agentProfileCellKey, s as assertRunAgentProfileCell, t as buildAgentInterfaceProfileCell, u as buildAgentProfileCell, v as groupRunsByAgentProfileCell, w as isRunRecord, x as modelHasSnapshot, y as parseRunRecordSafe, z as requireAgentProfileCell, B as roundTripRunRecord, C as toAgentProfileJson, D as validateAgentProfileCell, E as validateRunRecord, F as verifyAgentProfileCell } from './run-record-DksGsfgv.js';
|
|
4
|
+
import { B as BehavioralMetrics, y as RunScore, a as RunTrace, z as RunScoreWeights } from './semantic-concept-judge-D7z6JCLZ.js';
|
|
5
|
+
export { A as ConceptComplexity, E as ConceptFinding, G as ConceptSpec, H as ConceptWeightStrategy, C as CreateAnalystAiConfig, J as DEFAULT_COMPLEXITY_WEIGHTS, L as DEFAULT_RUN_SCORE_WEIGHTS, D as DEFAULT_TRACE_ANALYST_KINDS, c as DiffPolicy, F as FAILURE_MODE_KIND_SPEC, f as FindingSubject, g as FindingSubjectKind, i as FindingsDiff, j as FindingsStore, I as IMPROVEMENT_KIND_SPEC, k as KNOWLEDGE_GAP_KIND_SPEC, l as KNOWLEDGE_POISONING_KIND_SPEC, P as PersistedFinding, R as RunCritic, M as RunCriticOptions, N as SEMANTIC_CONCEPT_JUDGE_VERSION, m as SKILL_USAGE_ANALYST, b as SemanticConceptJudgeInput, S as SemanticConceptJudgeOptions, O as SemanticConceptJudgeResult, n as SkillUsageAnalyst, Q as SuboptimalCode, T as SuboptimalSignal, U as aggregateRunScore, V as clamp01, W as computeTraceMetrics, s as createAnalystAi, X as createSemanticConceptJudge, t as defaultIsMaterial, u as diffFindings, Y as runSemanticConceptJudge } from './semantic-concept-judge-D7z6JCLZ.js';
|
|
6
|
+
import { m as ChatRequest, q as CreateChatClientOpts } from './kind-factory-20hcaYpf.js';
|
|
7
|
+
export { a as Analyst, b as AnalystContext, i as AnalystCost, A as AnalystFinding, j as AnalystInputKind, k as AnalystRequirements, f as AnalystRunEvent, e as AnalystRunInputs, d as AnalystRunResult, c as AnalystRunSummary, g as AnalystSeverity, l as ChatCallOpts, C as ChatClient, n as ChatResponse, o as ChatTransport, p as CliBridgeTransportOpts, r as CreateTraceAnalystKindOpts, D as DirectProviderTransportOpts, E as EvidenceRef, M as MockTransportOpts, s as RawAnalystFinding, u as RouterTransportOpts, S as SandboxSdkTransportOpts, v as TraceAnalystGolden, T as TraceAnalystKindSpec, w as computeFindingId, x as createChatClient, y as createTraceAnalystKind, z as makeFinding, F as renderPriorFindings } from './kind-factory-20hcaYpf.js';
|
|
8
|
+
export { a as AnalystHooks, A as AnalystRegistry, b as AnalystRegistryOptions, B as BudgetPolicy, D as DefaultAnalystRegistryOptions, R as RegistryRunOpts, c as buildDefaultAnalystRegistry } from './default-registry-Bcf1uKVI.js';
|
|
9
|
+
export { F as FindingToPolicyEditOptions, P as POLICY_EDIT_AXES, a as POLICY_EDIT_TARGET_SURFACES, b as PolicyEdit, c as PolicyEditAdmission, d as PolicyEditAdmissionOptions, e as PolicyEditAxis, f as PolicyEditChange, g as PolicyEditExpectedGain, h as PolicyEditGainDirection, i as PolicyEditGainUnit, j as PolicyEditInit, k as PolicyEditRisk, l as PolicyEditSchemaVersion, m as PolicyEditSource, n as PolicyEditTarget, o as PolicyEditTargetSurface, p as PolicyEditValidationError, q as admitPolicyEdit, r as applyPolicyEditToSurface, s as computePolicyEditId, t as isPolicyEdit, u as makePolicyEdit, v as policyEditFromFinding, w as policyEditsFromFindings, x as scorePolicyEditReadiness, y as validatePolicyEdit } from './policy-edit-D2bBDZDf.js';
|
|
11
10
|
import { TCloud } from '@tangle-network/tcloud';
|
|
12
11
|
import { B as BenchmarkRunnerConfig, S as Scenario, c as BenchmarkReport, P as ProductClientConfig, C as CheckResult, T as TestResult, d as PersonaConfig, D as DriverResult, e as DriverState, b as JudgeFn, f as CollectedArtifacts, g as ScenarioResult, h as TurnMetrics, i as ScenarioFile, j as CompletionCriterion } from './types-C7DGg5ex.js';
|
|
13
12
|
export { A as ArtifactCheck, k as ArtifactResult, E as EvalResult, F as FeedbackPattern, l as JudgeConfig, a as JudgeInput, m as JudgeRubric, J as JudgeScore, n as PersonaRigor, R as RouteMap, o as RubricDimension, p as Turn, q as TurnResult } from './types-C7DGg5ex.js';
|
|
14
|
-
|
|
15
|
-
|
|
16
|
-
|
|
13
|
+
import { C as ControlSeverity, a as ControlEvalResult, b as FeedbackLabel, F as FeedbackTrajectoryStore, c as FeedbackTrajectory } from './feedback-trajectory-pDcz1lQ1.js';
|
|
14
|
+
export { d as ControlActionFailureMode, e as ControlActionOutcome, f as ControlBudget, g as ControlContext, h as ControlDecision, i as ControlRunResult, j as ControlRuntimeConfig, k as ControlRuntimeError, l as ControlStep, m as ControlStopPolicies, n as FeedbackArtifactType, o as FeedbackAttempt, p as FeedbackLabelKind, q as FeedbackLabelSource, r as FeedbackOptimizerRow, s as FeedbackOutcome, t as FeedbackReplayAdapter, u as FeedbackReplayResult, v as FeedbackSeverity, w as FeedbackSplitPolicy, x as FeedbackTask, y as FeedbackTrajectoryFilter, z as FileSystemFeedbackTrajectoryStore, I as InMemoryFeedbackTrajectoryStore, P as PreferenceMemoryEntry, A as ProposedSideEffect, S as StopDecision, B as allCriticalPassed, D as assignFeedbackSplit, E as controlRunToFeedbackTrajectory, G as createFeedbackTrajectory, H as feedbackTrajectoriesToDatasetScenarios, J as feedbackTrajectoriesToOptimizerRows, K as feedbackTrajectoryToDatasetScenario, L as feedbackTrajectoryToOptimizerRow, M as objectiveEval, N as parseFeedbackTrajectoriesJsonl, O as renderPreferenceMemoryMarkdown, Q as replayFeedbackTrajectories, R as replayFeedbackTrajectory, T as runAgentControlLoop, U as serializeFeedbackTrajectoriesJsonl, V as stopOnNoProgress, W as stopOnRepeatedAction, X as subjectiveEval, Y as summarizePreferenceMemory, Z as withAssignedFeedbackSplit } from './feedback-trajectory-pDcz1lQ1.js';
|
|
15
|
+
import { F as FailureClass, T as ToolSpan, h as BudgetSpec, B as BudgetLedgerEntry, R as Run, L as LlmSpan, S as Span } from './schema-SGWcK9wa.js';
|
|
16
|
+
export { A as Artifact, E as EventKind, i as FAILURE_CLASSES, G as GenericSpan, J as JudgeSpan, M as Message, c as RetrievalSpan, g as RunLayer, f as RunStatus, d as SandboxSpan, j as SpanBase, b as SpanKind, k as SpanStatus, l as TRACE_SCHEMA_VERSION, e as TraceEvent, m as isJudgeSpan, n as isLlmSpan, o as isRetrievalSpan, p as isSandboxSpan, q as isToolSpan } from './schema-SGWcK9wa.js';
|
|
17
17
|
import { A as AgentEvalError, J as JudgeError, a as ConfigError } from './errors-oeQrLqXC.js';
|
|
18
18
|
export { b as AgentEvalErrorCode, C as CaptureIntegrityError, N as NotFoundError, R as ReplayError, V as ValidationError, c as VerificationError } from './errors-oeQrLqXC.js';
|
|
19
|
-
import { b as
|
|
20
|
-
export {
|
|
21
|
-
import {
|
|
22
|
-
export {
|
|
23
|
-
|
|
24
|
-
|
|
25
|
-
export { A as ActionableSideInfo, o as AsiSeverity, B as BootstrapOptions, a as BootstrapResult, J as JudgeReplayGateArgs, R as ReleaseConfidenceAxis, b as ReleaseConfidenceAxisName, c as ReleaseConfidenceInput, d as ReleaseConfidenceIssue, e as ReleaseConfidenceMetrics, g as ReleaseConfidenceStatus, i as ReleaseTraceEvidence, j as RenderReleaseReportOptions, V as Verdict, k as assertReleaseConfidence, l as bootstrapCi, m as evaluateReleaseConfidence, n as judgeReplayGate, r as renderReleaseReport } from './release-report-DeJpsBiA.js';
|
|
19
|
+
import { b as CorrectnessChecker } from './pre-registration-BepVVa6P.js';
|
|
20
|
+
export { A as ArtifactCheckArtifact, d as ArtifactEventLike, e as ArtifactValidator, f as BackendIntegrityError, B as BackendIntegrityReport, C as CompletionRequirement, a as CompletionVerdict, H as HypothesisManifest, g as HypothesisResult, h as LlmCorrectnessCheckerOpts, L as LlmJudgeDimension, c as LlmJudgeOptions, i as ProducedProposal, P as ProducedState, j as ProposalEventLike, k as RequirementCheck, R as RuntimeEventLike, m as SatisfiedBy, S as SignedManifest, n as SignedManifestAlgo, T as TaskGold, o as ToolCallEventLike, V as ValidationContext, p as ValidationIssue, q as ValidationResult, r as assertRealBackend, s as byteLengthRange, t as canonicalize, u as completionVerdict, v as composeValidators, w as containsAll, x as createLlmCorrectnessChecker, y as createTokenRecallChecker, z as evaluateHypothesis, D as extractProducedState, E as hashJson, F as jsonHasKeys, l as llmJudge, G as parseCorrectnessResponse, I as regexMatch, J as signManifest, K as summarizeBackendIntegrity, M as verifyCompletion, N as verifyManifest } from './pre-registration-BepVVa6P.js';
|
|
21
|
+
import { T as TraceEmitter } from './emitter-BRchAAAx.js';
|
|
22
|
+
export { R as RunCompleteHook, a as RunCompleteHookContext, S as SpanHandle, b as TraceEmitterOptions, l as llmSpanFromProvider } from './emitter-BRchAAAx.js';
|
|
23
|
+
import { h as ReleaseConfidenceThresholds, f as ReleaseConfidenceScorecard } from './release-report-oBfOz8ku.js';
|
|
24
|
+
export { A as ActionableSideInfo, o as AsiSeverity, B as BootstrapOptions, a as BootstrapResult, J as JudgeReplayGateArgs, R as ReleaseConfidenceAxis, b as ReleaseConfidenceAxisName, c as ReleaseConfidenceInput, d as ReleaseConfidenceIssue, e as ReleaseConfidenceMetrics, g as ReleaseConfidenceStatus, i as ReleaseTraceEvidence, j as RenderReleaseReportOptions, V as Verdict, k as assertReleaseConfidence, l as bootstrapCi, m as evaluateReleaseConfidence, n as judgeReplayGate, r as renderReleaseReport } from './release-report-oBfOz8ku.js';
|
|
26
25
|
import { P as PairedBootstrapOptions, M as McNemarResult, R as RiskDifferenceResult, a as PairedBootstrapResult } from './statistics-D88peojY.js';
|
|
27
26
|
export { c as CliffsMagnitude, d as CorpusAgreementOptions, e as CorpusAgreementPerDimension, C as CorpusAgreementReport, f as CorpusScoreRecord, g as EProcess, h as EProcessOptions, E as EProcessState, i as EProcessStep, j as ProportionInterval, W as WeightedCompositeInput, k as WeightedCompositeResult, b as benjaminiHochberg, l as bonferroni, m as cliffsDelta, n as cohensD, o as confidenceInterval, q as corpusInterRaterAgreement, r as corpusInterRaterAgreementFromJudgeScores, s as eProcess, t as interRaterReliability, u as interpretCliffs, v as mannWhitneyU, x as mcnemar, y as mcnemarPower, z as mcnemarRequiredN, A as mulberry32, B as normalizeScores, p as pairedBootstrap, D as pairedMde, F as pairedRiskDifference, G as pairedTTest, H as partialCredit, I as passAtK, J as pearsonR, K as ranks, L as requiredSampleSize, N as spearmanR, O as weightedComposite, Q as weightedMean, w as wilcoxonSignedRank, S as wilson } from './statistics-D88peojY.js';
|
|
28
27
|
import { OtelExporter, OtelExportConfig } from './traces.js';
|
|
29
|
-
export { CaptureFetchContext, CaptureFetchOptions, ExportableSpan, ExtractedUsage, FlattenOtlpOptions, LLM_CACHED_TOKENS, LLM_CACHED_TOKEN_ATTR_KEYS, LLM_COST_ATTR_KEYS, LLM_COST_USD, LLM_INPUT_TOKENS, LLM_INPUT_TOKEN_ATTR_KEYS, LLM_MODEL_ATTR_KEYS, LLM_MODEL_NAME, LLM_OUTPUT_TOKENS, LLM_OUTPUT_TOKEN_ATTR_KEYS, OPENINFERENCE_SPAN_KIND, OTEL_AGENT_EVAL_SCOPE, OtlpExport, OtlpFileTraceStore, OtlpFileTraceStoreOptions, OtlpFlatLine, OtlpResourceSpans, OtlpSpan, OtlpToRunRecordsOptions, OtlpTraceRunRecord, ProjectedOtlpSpan, ReplayCache, ReplayCacheEntry, ReplayCacheMissError, ReplayCacheStats, ReplayFetchOptions, SPAN_KIND_ATTR_KEYS, SpanNotFoundError, TOOL_NAME, TOOL_NAME_ATTR_KEYS, TRACE_ANALYST_ACTOR_DESCRIPTION, TRACE_ANALYST_ACTOR_DESCRIPTION_VERSION, TRACE_ANALYST_SUBAGENT_DESCRIPTION, TraceAggregate, TraceAnalystHookOptions, TraceFileMissingError, TraceInsightContext, TraceInsightFinding, TraceInsightPanelRole, TraceInsightPromptInput, TraceInsightQualityGate, TraceInsightQuestion, TraceInsightReadiness, TraceInsightSuite, TraceInsightTask, TraceNotFoundError, TraceStoreSource, TraceStoreToOtlpOptions, TracesToOtlpResult, asNumber, asString, buildTraceAnalystTools, buildTraceInsightContext, buildTraceInsightPrompt, captureFetchToRawSink, convertTraceStoresToOtlp, createOtelExporter, createOtelTracingStore, createReplayFetch, defaultTraceInsightPanel, describeTraceInsightScope, domainEvidencePattern, exportRunAsOtlp, extractOtlpAttributes, extractUsage, extractUsageFromResponse, extractUsageFromSse, firstNumberAttr, firstStringAttr, flattenOtlpExportToNdjson, inferDomainKeywords, inferOtlpKind, iterateRawCalls, otelRunCompleteHook, otlpToRunRecords, otlpToTraceRunRecords, planTraceInsightQuestions, projectOtlpFlatLine, readOtlpStatus, scoreTraceInsightReadiness, stringField, tokenizeDomainWords, traceAnalystFunctionGroup, traceAnalystOnRunComplete, traceSpanKindToOpenInferenceKind } from './traces.js';
|
|
28
|
+
export { CaptureFetchContext, CaptureFetchOptions, DEFAULT_REDACTION_RULES, ExportableSpan, ExtractedUsage, FlattenOtlpOptions, LLM_CACHED_TOKENS, LLM_CACHED_TOKEN_ATTR_KEYS, LLM_COST_ATTR_KEYS, LLM_COST_USD, LLM_INPUT_TOKENS, LLM_INPUT_TOKEN_ATTR_KEYS, LLM_MODEL_ATTR_KEYS, LLM_MODEL_NAME, LLM_OUTPUT_TOKENS, LLM_OUTPUT_TOKEN_ATTR_KEYS, OPENINFERENCE_SPAN_KIND, OTEL_AGENT_EVAL_SCOPE, OtlpExport, OtlpFileTraceStore, OtlpFileTraceStoreOptions, OtlpFlatLine, OtlpResourceSpans, OtlpSpan, OtlpToRunRecordsOptions, OtlpTraceRunRecord, ProjectedOtlpSpan, REDACTION_VERSION, RedactionReport, RedactionRule, ReplayCache, ReplayCacheEntry, ReplayCacheMissError, ReplayCacheStats, ReplayFetchOptions, SPAN_KIND_ATTR_KEYS, SpanNotFoundError, TOOL_NAME, TOOL_NAME_ATTR_KEYS, TRACE_ANALYST_ACTOR_DESCRIPTION, TRACE_ANALYST_ACTOR_DESCRIPTION_VERSION, TRACE_ANALYST_SUBAGENT_DESCRIPTION, TraceAggregate, TraceAnalystHookOptions, TraceFileMissingError, TraceInsightContext, TraceInsightFinding, TraceInsightPanelRole, TraceInsightPromptInput, TraceInsightQualityGate, TraceInsightQuestion, TraceInsightReadiness, TraceInsightSuite, TraceInsightTask, TraceNotFoundError, TraceStoreSource, TraceStoreToOtlpOptions, TracesToOtlpResult, asNumber, asString, buildTraceAnalystTools, buildTraceInsightContext, buildTraceInsightPrompt, captureFetchToRawSink, convertTraceStoresToOtlp, createOtelExporter, createOtelTracingStore, createReplayFetch, defaultTraceInsightPanel, describeTraceInsightScope, domainEvidencePattern, exportRunAsOtlp, extractOtlpAttributes, extractUsage, extractUsageFromResponse, extractUsageFromSse, firstNumberAttr, firstStringAttr, flattenOtlpExportToNdjson, inferDomainKeywords, inferOtlpKind, iterateRawCalls, otelRunCompleteHook, otlpToRunRecords, otlpToTraceRunRecords, planTraceInsightQuestions, projectOtlpFlatLine, readOtlpStatus, redactString, redactValue, scoreTraceInsightReadiness, stringField, tokenizeDomainWords, traceAnalystFunctionGroup, traceAnalystOnRunComplete, traceSpanKindToOpenInferenceKind } from './traces.js';
|
|
30
29
|
import { a as AnalyzeTracesInput, A as AnalyzeTracesOptions, b as AnalyzeTracesResult } from './analyst-C8HHvfJp.js';
|
|
31
30
|
export { c as AnalyzeTracesTurnSnapshot, d as analyzeTraces } from './analyst-C8HHvfJp.js';
|
|
32
31
|
import { a as TraceAnalystSpan } from './store-C1YxJDEK.js';
|
|
33
32
|
export { D as DEFAULT_TRACE_ANALYST_BUDGETS, b as DatasetOverview, E as ErrorCluster, Q as QueryTracesPage, S as SearchSpanResult, c as SearchTraceResult, d as SpanMatchRecord, e as TRACE_ANALYST_TRUNCATION_MARKER_PREFIX, T as TraceAnalysisStore, f as TraceAnalystByteBudgets, g as TraceAnalystFilters, h as TraceAnalystSpanKind, i as TraceAnalystSpanStatus, j as TraceAnalystTraceSummary, V as ViewSpansResult, k as ViewTraceOversized, l as ViewTraceResult } from './store-C1YxJDEK.js';
|
|
34
|
-
import {
|
|
35
|
-
import { A as AnalyzeRunsOptions } from './analyze-runs-
|
|
36
|
-
import {
|
|
37
|
-
export {
|
|
38
|
-
import { S as SandboxDriver, H as HarnessConfig, a as SandboxHarnessResult } from './test-graded-scenario-
|
|
39
|
-
export { D as DockerSandboxDriver, c as SandboxHarness, d as SandboxResult, e as SubprocessSandboxDriver, f as SubprocessSandboxDriverOptions, g as TestGradedRunOptions, b as TestGradedRunResult, T as TestGradedScenario, h as TestOutputParser, i as composeParsers, j as jestTestParser, p as pytestTestParser, r as runTestGradedScenario, v as vitestTestParser } from './test-graded-scenario-
|
|
40
|
-
|
|
41
|
-
export {
|
|
42
|
-
import { T as TraceEmitter } from './emitter-C2rqGH_l.js';
|
|
43
|
-
export { R as RunCompleteHook, a as RunCompleteHookContext, S as SpanHandle, b as TraceEmitterOptions, l as llmSpanFromProvider } from './emitter-C2rqGH_l.js';
|
|
44
|
-
export { b as RunIntegrityError, R as RunIntegrityExpectations, c as RunIntegrityIssue, d as RunIntegrityIssueCode, a as RunIntegrityReport, e as assertRunCaptured, t as throwIfRunIncomplete } from './integrity-DqGZg3st.js';
|
|
45
|
-
export { a as aggregateLlm, b as argHash, g as groupBy, j as judgeSpans, l as llmSpans, r as runFailureClass, c as runsForScenario, t as toolSpans } from './query-0aTmbmQe.js';
|
|
33
|
+
import { b as JudgeConfig, J as JudgeScore, S as Scenario$1, g as Gate } from './types-v--ctu-b.js';
|
|
34
|
+
import { A as AnalyzeRunsOptions } from './analyze-runs-Dmz6LA9e.js';
|
|
35
|
+
import { q as Objective, t as ParetoResult, h as GepaProposerConstraints, a as RunImprovementLoopResult } from './gepa-BUNP3606.js';
|
|
36
|
+
export { u as DEFAULT_RED_TEAM_CORPUS, D as Direction, e as RedTeamCase, v as RedTeamCategory, w as RedTeamFinding, x as RedTeamPayload, y as RedTeamReport, z as crowdingDistance, A as dominates, B as paretoFrontier, E as paretoFrontierWithCrowding, F as redTeamDataset, H as redTeamReport, I as scalarScore, J as scoreRedTeamOutput, K as toolNamesForRun } from './gepa-BUNP3606.js';
|
|
37
|
+
import { S as SandboxDriver, H as HarnessConfig, a as SandboxHarnessResult } from './test-graded-scenario-mzYBKspu.js';
|
|
38
|
+
export { D as DockerSandboxDriver, c as SandboxHarness, d as SandboxResult, e as SubprocessSandboxDriver, f as SubprocessSandboxDriverOptions, g as TestGradedRunOptions, b as TestGradedRunResult, T as TestGradedScenario, h as TestOutputParser, i as composeParsers, j as jestTestParser, p as pytestTestParser, r as runTestGradedScenario, v as vitestTestParser } from './test-graded-scenario-mzYBKspu.js';
|
|
39
|
+
export { b as RunIntegrityError, R as RunIntegrityExpectations, c as RunIntegrityIssue, d as RunIntegrityIssueCode, a as RunIntegrityReport, e as assertRunCaptured, t as throwIfRunIncomplete } from './integrity-qemeBAyx.js';
|
|
40
|
+
export { a as aggregateLlm, b as argHash, g as groupBy, j as judgeSpans, l as llmSpans, r as runFailureClass, c as runsForScenario, t as toolSpans } from './query-Ck190MOd.js';
|
|
46
41
|
export { F as FileSystemRawProviderSink, a as FileSystemRawProviderSinkOptions, I as InMemoryRawProviderSink, b as InMemoryRawProviderSinkOptions, N as NoopRawProviderSink, P as ProviderRedactor, c as RawProviderDirection, d as RawProviderEvent, R as RawProviderSink, e as RawProviderSinkFilter, f as defaultProviderRedactor, p as providerFromBaseUrl } from './raw-provider-sink-C46HDghv.js';
|
|
47
|
-
|
|
48
|
-
|
|
49
|
-
export {
|
|
50
|
-
export {
|
|
51
|
-
|
|
52
|
-
|
|
53
|
-
export { B as BaselineOptions, M as MetricSamples, b as MetricVerdict, T as ToolStats, d as ToolUseMetrics, e as ToolUseOptions, f as compareToBaseline, c as computeToolUseMetrics, i as iqr, w as welchsTTest } from './baseline-Bbid3WoO.js';
|
|
54
|
-
import { a as TrajectoryStep, T as Trajectory } from './trajectory-2TkpSEVh.js';
|
|
55
|
-
export { b as buildTrajectory } from './trajectory-2TkpSEVh.js';
|
|
42
|
+
import { T as TraceStore, R as RunFilter } from './store-BsVi7ncX.js';
|
|
43
|
+
export { E as EventFilter, F as FileSystemTraceStore, a as FileSystemTraceStoreOptions, I as InMemoryTraceStore, S as SpanFilter } from './store-BsVi7ncX.js';
|
|
44
|
+
export { D as DEFAULT_FAILURE_RULES, b as FailureClassification, c as FailureContext, d as FailureRule, e as classifyFailure } from './failure-cluster-C48PiReX.js';
|
|
45
|
+
export { P as ProjectRuntimeTrajectoryEvidenceOptions, a as RuntimeTrajectoryEvidenceProjection, b as RuntimeTrajectoryEvidenceSummary, c as RuntimeTrajectoryHookEvent, R as RuntimeTrajectoryRecord, d as RuntimeTrajectoryRunRecord, p as parseRuntimeTrajectoryHookEvent, e as projectRuntimeTrajectoryEvidence } from './runtime-trajectory-h5i0SZUj.js';
|
|
46
|
+
import { a as BaselineReport, b as Trajectory, T as TrajectoryStep } from './baseline-DsNteOgR.js';
|
|
47
|
+
export { B as BaselineOptions, M as MetricSamples, d as MetricVerdict, e as ToolStats, f as ToolUseMetrics, g as ToolUseOptions, h as buildTrajectory, i as compareToBaseline, c as computeToolUseMetrics, j as iqr, w as welchsTTest } from './baseline-DsNteOgR.js';
|
|
56
48
|
import { HarnessType, AgentProfile } from '@tangle-network/agent-interface';
|
|
57
49
|
export { AgentProfile, HarnessType } from '@tangle-network/agent-interface';
|
|
58
50
|
import { b as ChannelRollup, C as CostLedger } from './cost-ledger-DuSqlw5B.js';
|
|
59
51
|
export { a as CostChannel, c as CostLedgerEntry, d as CostLedgerSummary, e as CostResult, f as CostUsage, g as costForUsage, m as modelPriceKey } from './cost-ledger-DuSqlw5B.js';
|
|
60
|
-
export { D as Direction, O as Objective, P as ParetoResult, c as crowdingDistance, d as dominates, p as paretoFrontier, a as paretoFrontierWithCrowding, s as scalarScore } from './pareto-E-pembql.js';
|
|
61
52
|
export { S as SeriesConvergenceOptions, a as SeriesConvergenceResult, b as analyzeSeries } from './series-convergence-D5OWMBg6.js';
|
|
62
53
|
import { D as DefaultVerdict } from './verdict-C9MlYujm.js';
|
|
63
|
-
import {
|
|
64
|
-
export { d as DatasetDifficulty,
|
|
54
|
+
import { D as DatasetScenario, c as Dataset } from './dataset-NENEzRgk.js';
|
|
55
|
+
export { d as DatasetDifficulty, b as DatasetManifest, e as DatasetProvenance, a as DatasetSplit, H as HoldoutLockedError, S as SliceOptions, h as hashScenarios } from './dataset-NENEzRgk.js';
|
|
65
56
|
export { a as CalibrationResult, c as CandidateScore, C as ContinuousAgreement, d as ContinuousAgreementOptions, b as ContinuousCalibrationResult, G as GoldenItem, P as PositionalBiasResult, S as SelfPreferenceResult, V as VerbosityBiasResult, e as calibrateJudge, f as calibrateJudgeContinuous, g as continuousAgreement, p as positionalBias, s as selfPreference, v as verbosityBias } from './judge-calibration-7C-IDmKr.js';
|
|
66
|
-
|
|
67
|
-
export {
|
|
68
|
-
import { a as PrmGrader } from './rubric-Cc6UHvUb.js';
|
|
69
|
-
export { EuRiskClass, GovernanceContext, GovernanceFinding, GovernanceReport, UseCaseSignals, classifyEuAiRisk, euAiActReport, nistAiRmfReport, renderMarkdown, soc2Report, summarize } from './governance/index.js';
|
|
70
|
-
import { b as Layer, S as Severity, L as LayerResult, c as VerifyContext } from './multi-layer-verifier-CI4jdX-q.js';
|
|
71
|
-
export { F as Finding, d as LayerStatus, M as MultiLayerVerifier, a as VerificationReport, V as VerifyOptions, g as gradeSemanticStatus } from './multi-layer-verifier-CI4jdX-q.js';
|
|
57
|
+
import { L as Layer, S as Severity, b as LayerResult, c as VerifyContext } from './multi-layer-verifier-BsqKuLyN.js';
|
|
58
|
+
export { F as Finding, d as LayerStatus, M as MultiLayerVerifier, a as VerificationReport, V as VerifyOptions, g as gradeSemanticStatus } from './multi-layer-verifier-BsqKuLyN.js';
|
|
72
59
|
import { L as LlmClientOptions } from './llm-client-DyqEH4jH.js';
|
|
73
60
|
export { d as LlmCallError, b as LlmCallRequest, c as LlmCallResult, e as LlmClient, f as LlmMessage, g as LlmRouteAssertionError, a as LlmRouteRequirements, h as LlmUsage, i as assertLlmRoute, j as backoffMs, k as callLlm, l as callLlmJson, m as isTransientLlmError, p as probeLlm, s as stripFencedJson } from './llm-client-DyqEH4jH.js';
|
|
74
|
-
export { B as BENCHMARK_SPLIT_SEED, a as BenchmarkAdapter, b as BenchmarkDatasetItem, c as BenchmarkEvaluation, d as BenchmarkFamily, e as BenchmarkResponder, f as BenchmarkScenario, g as BenchmarkSource, h as BenchmarkTaskKind, i as benchmarkDeterministicSplit, j as benchmarks } from './index-
|
|
75
|
-
export { C as CallbackResearcher, d as CallbackResearcherOptions, e as CampaignFactoryParams, f as CampaignIntegrityPolicy, g as CampaignRunContext, h as CampaignRunOutcome, i as CampaignRunner, j as CampaignScenario, k as CampaignVariant, c as EvalCampaignOptions, b as EvalCampaignResult, E as ExperimentPlan, a as ExperimentResult, l as FailedRun, F as FailureMode, N as NoopResearcher, R as Researcher, S as SteeringChange, r as runEvalCampaign } from './researcher-
|
|
76
|
-
export { G as GainDistributionBin, a as GainDistributionFigureSpec, b as GainDistributionOptions, m as GateDecision, n as GateEvidence, H as HeldOutGate, o as HeldOutGateConfig, q as HeldOutGateRejectionCode, P as ParetoFigureSpec, c as ParetoPoint, R as RESEARCH_REPORT_HARD_PAIR_FLOOR, d as ResearchReport, e as ResearchReportCandidate, f as ResearchReportDecision, g as ResearchReportMethodology, h as ResearchReportOptions, i as ResearchReportRecommendation, S as SummaryTable, j as SummaryTableOptions, k as SummaryTableRow, l as gainHistogram, p as paretoChart, r as researchReport, s as summaryTable } from './summary-report-
|
|
77
|
-
export { L as LockedJsonlAppender } from './testing-C21CHsq2.js';
|
|
61
|
+
export { B as BENCHMARK_SPLIT_SEED, a as BenchmarkAdapter, b as BenchmarkDatasetItem, c as BenchmarkEvaluation, d as BenchmarkFamily, e as BenchmarkResponder, f as BenchmarkScenario, g as BenchmarkSource, h as BenchmarkTaskKind, i as benchmarkDeterministicSplit, j as benchmarks } from './index-Dc3VLGhp.js';
|
|
62
|
+
export { C as CallbackResearcher, d as CallbackResearcherOptions, e as CampaignFactoryParams, f as CampaignIntegrityPolicy, g as CampaignRunContext, h as CampaignRunOutcome, i as CampaignRunner, j as CampaignScenario, k as CampaignVariant, c as EvalCampaignOptions, b as EvalCampaignResult, E as ExperimentPlan, a as ExperimentResult, l as FailedRun, F as FailureMode, N as NoopResearcher, R as Researcher, S as SteeringChange, r as runEvalCampaign } from './researcher-CaH0CwFC.js';
|
|
63
|
+
export { G as GainDistributionBin, a as GainDistributionFigureSpec, b as GainDistributionOptions, m as GateDecision, n as GateEvidence, H as HeldOutGate, o as HeldOutGateConfig, q as HeldOutGateRejectionCode, P as ParetoFigureSpec, c as ParetoPoint, R as RESEARCH_REPORT_HARD_PAIR_FLOOR, d as ResearchReport, e as ResearchReportCandidate, f as ResearchReportDecision, g as ResearchReportMethodology, h as ResearchReportOptions, i as ResearchReportRecommendation, S as SummaryTable, j as SummaryTableOptions, k as SummaryTableRow, l as gainHistogram, p as paretoChart, r as researchReport, s as summaryTable } from './summary-report-Bz-0-t8v.js';
|
|
78
64
|
export { I as InterimReleaseConfidence, a as InterimReleaseConfidenceInput, P as PairedEvalueOptions, b as PairedEvalueSequence, c as PairedEvalueStep, S as SequentialDecision, e as evaluateInterimReleaseConfidence, p as pairedEvalueSequence } from './sequential-5iSVfzl2.js';
|
|
79
|
-
import { f as GepaProposerConstraints, a as RunImprovementLoopResult } from './gepa-B3x5Ulcv.js';
|
|
80
|
-
export { IntegrityResult, IntegrityViolation, JourneySpec, PerfBaseline, PerfGateResult, PerfRegression, PerfScenario, PerfStat, ScenarioAxes, assertRecordIntegrity, checkRecordIntegrity, expandMatrix, gatePerf, scenarioKey, summarizeRecords } from './perf/index.js';
|
|
81
|
-
export { AgentProfileRuntimeReceipt, ProductBenchmarkArm, ProductBenchmarkArtifactPaths, ProductBenchmarkBudgets, ProductBenchmarkExportOptions, ProductBenchmarkExportResult, ProductBenchmarkManifest, ProductBenchmarkProfileRef, ProductBenchmarkRecord, ProductBenchmarkRepoRef, ProductBenchmarkRunInput, ProductBenchmarkScenario, ProductBenchmarkSingleRunExportOptions, ProductBenchmarkSplit, ProductBenchmarkSubstrateVersions, ProductBenchmarkValidationReport, RuntimeResolution, assertProductBenchmarkRun, buildProductBenchmarkManifest, exportProductBenchmark, exportProductBenchmarkRuns, findProductBenchmarkArtifacts, productBenchmarkIntegrityFailures, productBenchmarkMutableSurfaces, productBenchmarkRepoIdentity, productBenchmarkSplits, readProductBenchmarkManifest, readProductBenchmarkRecords, runRecordToProductBenchmarkRecord, validateProductBenchmarkManifest, validateProductBenchmarkRecord, validateProductBenchmarkRun } from './product-benchmark/index.js';
|
|
82
65
|
import '@ax-llm/ax';
|
|
83
66
|
import 'zod';
|
|
84
|
-
import './insight-report-
|
|
85
|
-
import './outcome-store-rnXLEqSn.js';
|
|
67
|
+
import './insight-report-D4cXFsLt.js';
|
|
86
68
|
import './storage-Dw_f7WMt.js';
|
|
87
69
|
|
|
88
70
|
/**
|
|
@@ -839,6 +821,103 @@ declare function createCustomJudge(name: string, systemPrompt: string, opts?: {
|
|
|
839
821
|
*/
|
|
840
822
|
declare function defaultJudges(domain: string): JudgeFn[];
|
|
841
823
|
|
|
824
|
+
type KnowledgeRequirementCategory = 'user_specific' | 'company_specific' | 'domain_specific' | 'codebase_specific' | 'market_specific' | 'regulatory' | 'tool_api' | 'credential_or_secret' | 'runtime_environment' | 'preference' | 'historical_context';
|
|
825
|
+
type KnowledgeAcquisitionMode = 'ask_user' | 'search_web' | 'query_connector' | 'inspect_repo' | 'run_command' | 'infer_low_confidence' | 'not_available';
|
|
826
|
+
type KnowledgeImportance = 'blocking' | 'high' | 'medium' | 'low';
|
|
827
|
+
type KnowledgeFreshness = 'static' | 'monthly' | 'weekly' | 'daily' | 'realtime';
|
|
828
|
+
type KnowledgeSensitivity = 'public' | 'private' | 'secret';
|
|
829
|
+
type KnowledgeFallbackPolicy = 'block' | 'ask' | 'continue_with_caveat' | 'use_default';
|
|
830
|
+
interface KnowledgeRequirement {
|
|
831
|
+
id: string;
|
|
832
|
+
description: string;
|
|
833
|
+
requiredFor: string[];
|
|
834
|
+
category: KnowledgeRequirementCategory;
|
|
835
|
+
acquisitionMode: KnowledgeAcquisitionMode;
|
|
836
|
+
importance: KnowledgeImportance;
|
|
837
|
+
freshness: KnowledgeFreshness;
|
|
838
|
+
sensitivity: KnowledgeSensitivity;
|
|
839
|
+
confidenceNeeded: number;
|
|
840
|
+
currentConfidence: number;
|
|
841
|
+
evidenceIds: string[];
|
|
842
|
+
fallbackPolicy: KnowledgeFallbackPolicy;
|
|
843
|
+
/**
|
|
844
|
+
* ISO timestamp after which this requirement must be treated as stale.
|
|
845
|
+
* Stale requirements score as missing even when they still have evidence.
|
|
846
|
+
*/
|
|
847
|
+
validUntil?: string;
|
|
848
|
+
/** ISO timestamp for the last source-grounding or human verification pass. */
|
|
849
|
+
lastVerifiedAt?: string;
|
|
850
|
+
metadata?: Record<string, unknown>;
|
|
851
|
+
}
|
|
852
|
+
interface KnowledgeBundle {
|
|
853
|
+
taskId: string;
|
|
854
|
+
requirements: KnowledgeRequirement[];
|
|
855
|
+
evidenceIds: string[];
|
|
856
|
+
claimIds: string[];
|
|
857
|
+
wikiPageIds: string[];
|
|
858
|
+
userAnswers: Record<string, string>;
|
|
859
|
+
missing: KnowledgeRequirement[];
|
|
860
|
+
readinessScore: number;
|
|
861
|
+
metadata?: Record<string, unknown>;
|
|
862
|
+
}
|
|
863
|
+
type KnowledgeRecommendedAction = 'run_agent' | 'ask_user' | 'collect_web_data' | 'query_connectors' | 'inspect_repo' | 'build_domain_wiki' | 'continue_with_caveat' | 'abort_or_rescope';
|
|
864
|
+
interface KnowledgeReadinessReport {
|
|
865
|
+
taskId: string;
|
|
866
|
+
readinessScore: number;
|
|
867
|
+
blockingMissingRequirements: KnowledgeRequirement[];
|
|
868
|
+
nonBlockingGaps: KnowledgeRequirement[];
|
|
869
|
+
recommendedAction: KnowledgeRecommendedAction;
|
|
870
|
+
bundle: KnowledgeBundle;
|
|
871
|
+
severity: ControlSeverity;
|
|
872
|
+
reason: string;
|
|
873
|
+
}
|
|
874
|
+
interface UserQuestion {
|
|
875
|
+
id: string;
|
|
876
|
+
question: string;
|
|
877
|
+
reason: string;
|
|
878
|
+
requirementId: string;
|
|
879
|
+
importance: KnowledgeImportance;
|
|
880
|
+
answerType: 'free_text' | 'select_one' | 'multi_select' | 'file_upload' | 'credential' | 'url';
|
|
881
|
+
defaultIfSkipped?: string;
|
|
882
|
+
impactIfUnknown: string;
|
|
883
|
+
options?: string[];
|
|
884
|
+
metadata?: Record<string, unknown>;
|
|
885
|
+
}
|
|
886
|
+
interface DataAcquisitionPlan {
|
|
887
|
+
id: string;
|
|
888
|
+
requirementIds: string[];
|
|
889
|
+
mode: Exclude<KnowledgeAcquisitionMode, 'not_available' | 'infer_low_confidence'> | 'build_domain_wiki';
|
|
890
|
+
description: string;
|
|
891
|
+
priority: KnowledgeImportance;
|
|
892
|
+
expectedEvidenceIds?: string[];
|
|
893
|
+
questions?: UserQuestion[];
|
|
894
|
+
metadata?: Record<string, unknown>;
|
|
895
|
+
}
|
|
896
|
+
type KnowledgeResponsibleSurface = 'knowledge-requirements' | 'data-acquisition' | 'retrieval-policy' | 'user-question-policy';
|
|
897
|
+
|
|
898
|
+
interface ScoreKnowledgeReadinessOptions {
|
|
899
|
+
taskId: string;
|
|
900
|
+
requirements: KnowledgeRequirement[];
|
|
901
|
+
evidenceIds?: string[];
|
|
902
|
+
claimIds?: string[];
|
|
903
|
+
wikiPageIds?: string[];
|
|
904
|
+
userAnswers?: Record<string, string>;
|
|
905
|
+
metadata?: Record<string, unknown>;
|
|
906
|
+
now?: Date;
|
|
907
|
+
}
|
|
908
|
+
declare function scoreKnowledgeReadiness(options: ScoreKnowledgeReadinessOptions): KnowledgeReadinessReport;
|
|
909
|
+
declare function blockingKnowledgeEval(report: KnowledgeReadinessReport, options?: {
|
|
910
|
+
id?: string;
|
|
911
|
+
minimumScore?: number;
|
|
912
|
+
emitter?: TraceEmitter;
|
|
913
|
+
}): ControlEvalResult;
|
|
914
|
+
declare function knowledgeReadinessTracePayload(report: KnowledgeReadinessReport, options?: {
|
|
915
|
+
passed?: boolean;
|
|
916
|
+
minimumScore?: number;
|
|
917
|
+
}): Record<string, unknown>;
|
|
918
|
+
declare function userQuestionsForKnowledgeGaps(gaps: KnowledgeRequirement[]): UserQuestion[];
|
|
919
|
+
declare function acquisitionPlansForKnowledgeGaps(gaps: KnowledgeRequirement[]): DataAcquisitionPlan[];
|
|
920
|
+
|
|
842
921
|
interface LiveProofArtifact {
|
|
843
922
|
kind: string;
|
|
844
923
|
id?: string;
|
|
@@ -1572,6 +1651,108 @@ declare function toOpenAiTool(def: EvalToolDef): {
|
|
|
1572
1651
|
*/
|
|
1573
1652
|
declare function makeEvalTools(cfg: MakeEvalToolsConfig): EvalToolDef[];
|
|
1574
1653
|
|
|
1654
|
+
interface SteeringRolePrompt {
|
|
1655
|
+
system?: string;
|
|
1656
|
+
append?: string;
|
|
1657
|
+
}
|
|
1658
|
+
interface SteeringBundle {
|
|
1659
|
+
id: string;
|
|
1660
|
+
coderPrompt?: string;
|
|
1661
|
+
continuePrompt?: string;
|
|
1662
|
+
reviewerPrompts?: Record<string, string>;
|
|
1663
|
+
skills?: string[];
|
|
1664
|
+
rolePrompts?: Record<string, SteeringRolePrompt>;
|
|
1665
|
+
metadata?: Record<string, unknown>;
|
|
1666
|
+
}
|
|
1667
|
+
interface SteeringDelta {
|
|
1668
|
+
coderPrompt?: string;
|
|
1669
|
+
continuePrompt?: string;
|
|
1670
|
+
reviewerPrompts?: Record<string, string>;
|
|
1671
|
+
skills?: string[];
|
|
1672
|
+
rolePrompts?: Record<string, SteeringRolePrompt>;
|
|
1673
|
+
metadata?: Record<string, unknown>;
|
|
1674
|
+
}
|
|
1675
|
+
declare function mergeSteeringBundle(base: SteeringBundle, delta: SteeringDelta): SteeringBundle;
|
|
1676
|
+
declare function renderSteeringText(bundle: SteeringBundle): string;
|
|
1677
|
+
|
|
1678
|
+
type HarnessIntervention = 'continue' | 'plan' | 'audit' | 'recover' | 'repair' | 'verify' | 'final_gate' | 'wait_for_measurement' | 'abort';
|
|
1679
|
+
interface WorkflowTopology {
|
|
1680
|
+
id: string;
|
|
1681
|
+
interventions: HarnessIntervention[];
|
|
1682
|
+
maxParallelBranches?: number;
|
|
1683
|
+
metadata?: Record<string, unknown>;
|
|
1684
|
+
}
|
|
1685
|
+
interface MeasurementPolicy {
|
|
1686
|
+
required: string[];
|
|
1687
|
+
optional?: string[];
|
|
1688
|
+
promoteOn?: Array<keyof RunScore | 'aggregate'>;
|
|
1689
|
+
}
|
|
1690
|
+
interface HarnessVariant {
|
|
1691
|
+
id: string;
|
|
1692
|
+
steering?: SteeringBundle;
|
|
1693
|
+
topology?: WorkflowTopology;
|
|
1694
|
+
measurement?: MeasurementPolicy;
|
|
1695
|
+
budgets?: Record<string, number>;
|
|
1696
|
+
models?: Record<string, string>;
|
|
1697
|
+
reviewers?: Record<string, string>;
|
|
1698
|
+
metadata?: Record<string, unknown>;
|
|
1699
|
+
}
|
|
1700
|
+
interface HarnessScenario {
|
|
1701
|
+
id: string;
|
|
1702
|
+
task: string;
|
|
1703
|
+
split?: 'train' | 'validation' | 'test' | string;
|
|
1704
|
+
metadata?: Record<string, unknown>;
|
|
1705
|
+
}
|
|
1706
|
+
interface HarnessRunRequest {
|
|
1707
|
+
variant: HarnessVariant;
|
|
1708
|
+
scenario: HarnessScenario;
|
|
1709
|
+
trialIndex: number;
|
|
1710
|
+
}
|
|
1711
|
+
interface HarnessAdapter {
|
|
1712
|
+
run(request: HarnessRunRequest): Promise<RunTrace>;
|
|
1713
|
+
}
|
|
1714
|
+
interface HarnessRunResult {
|
|
1715
|
+
variant: HarnessVariant;
|
|
1716
|
+
scenario: HarnessScenario;
|
|
1717
|
+
trialIndex: number;
|
|
1718
|
+
trace: RunTrace;
|
|
1719
|
+
score: RunScore;
|
|
1720
|
+
aggregate: number;
|
|
1721
|
+
}
|
|
1722
|
+
interface HarnessVariantReport {
|
|
1723
|
+
variant: HarnessVariant;
|
|
1724
|
+
runs: HarnessRunResult[];
|
|
1725
|
+
aggregateMean: number;
|
|
1726
|
+
passRate: number;
|
|
1727
|
+
costUsdMean: number;
|
|
1728
|
+
wallSecondsMean: number;
|
|
1729
|
+
scoreMean: RunScore;
|
|
1730
|
+
}
|
|
1731
|
+
interface HarnessSelection {
|
|
1732
|
+
winner: HarnessVariantReport;
|
|
1733
|
+
frontier: ParetoResult<HarnessVariantReport>;
|
|
1734
|
+
reports: HarnessVariantReport[];
|
|
1735
|
+
}
|
|
1736
|
+
interface HarnessExperimentResult {
|
|
1737
|
+
results: HarnessRunResult[];
|
|
1738
|
+
selection: HarnessSelection;
|
|
1739
|
+
}
|
|
1740
|
+
interface HarnessExperimentConfig {
|
|
1741
|
+
adapter: HarnessAdapter;
|
|
1742
|
+
variants: HarnessVariant[];
|
|
1743
|
+
scenarios: HarnessScenario[];
|
|
1744
|
+
trialsPerScenario?: number;
|
|
1745
|
+
parallelism?: number;
|
|
1746
|
+
weights?: Partial<RunScoreWeights>;
|
|
1747
|
+
objectives?: Array<Objective<HarnessVariantReport>>;
|
|
1748
|
+
score?: (trace: RunTrace, request: HarnessRunRequest) => RunScore | Promise<RunScore>;
|
|
1749
|
+
onResult?: (result: HarnessRunResult) => void | Promise<void>;
|
|
1750
|
+
}
|
|
1751
|
+
declare const DEFAULT_HARNESS_OBJECTIVES: Array<Objective<HarnessVariantReport>>;
|
|
1752
|
+
declare function runHarnessExperiment(config: HarnessExperimentConfig): Promise<HarnessExperimentResult>;
|
|
1753
|
+
declare function selectHarnessVariant(results: HarnessRunResult[], objectives?: Array<Objective<HarnessVariantReport>>): HarnessSelection;
|
|
1754
|
+
declare function summarizeHarnessResults(results: HarnessRunResult[]): HarnessVariantReport[];
|
|
1755
|
+
|
|
1575
1756
|
/**
|
|
1576
1757
|
* Judge-ensemble reducer — folds N independent judge verdicts on the same
|
|
1577
1758
|
* artifact into one aggregate score.
|
|
@@ -3929,6 +4110,86 @@ declare function promptBisect(options: {
|
|
|
3929
4110
|
offendingParagraphIndex?: number;
|
|
3930
4111
|
}>;
|
|
3931
4112
|
|
|
4113
|
+
/**
|
|
4114
|
+
* Counterfactual replay — "what would have happened if we'd changed
|
|
4115
|
+
* exactly one thing at turn N?"
|
|
4116
|
+
*
|
|
4117
|
+
* The framework does NOT drive the agent — it sets up the replay
|
|
4118
|
+
* context (prior spans, prior state, mutation spec) and records the
|
|
4119
|
+
* resulting divergence. Consumers supply an `executeFrom(ctx)` callback
|
|
4120
|
+
* that runs their agent starting from turn N with the mutation applied.
|
|
4121
|
+
*
|
|
4122
|
+
* Counterfactual runs are recorded as a new Run with `layer='meta'` and
|
|
4123
|
+
* `parentRunId = originalRunId`, so downstream diff + correlation
|
|
4124
|
+
* pipelines see them natively.
|
|
4125
|
+
*/
|
|
4126
|
+
|
|
4127
|
+
type CounterfactualMutation = {
|
|
4128
|
+
kind: 'swap-model';
|
|
4129
|
+
at: number;
|
|
4130
|
+
newModel: string;
|
|
4131
|
+
} | {
|
|
4132
|
+
kind: 'swap-tool-result';
|
|
4133
|
+
at: number;
|
|
4134
|
+
newResult: unknown;
|
|
4135
|
+
} | {
|
|
4136
|
+
kind: 'truncate-after';
|
|
4137
|
+
at: number;
|
|
4138
|
+
} | {
|
|
4139
|
+
kind: 'inject-system-message';
|
|
4140
|
+
at: number;
|
|
4141
|
+
content: string;
|
|
4142
|
+
} | {
|
|
4143
|
+
kind: 'custom';
|
|
4144
|
+
at: number;
|
|
4145
|
+
describe: string;
|
|
4146
|
+
apply: (step: TrajectoryStep) => TrajectoryStep;
|
|
4147
|
+
};
|
|
4148
|
+
interface CounterfactualContext {
|
|
4149
|
+
originalRunId: string;
|
|
4150
|
+
originalTrajectory: Trajectory;
|
|
4151
|
+
/** Steps up to (but not including) the mutation point — the prefix the
|
|
4152
|
+
* replayed agent inherits as its prior conversation/tool history. */
|
|
4153
|
+
prefix: TrajectoryStep[];
|
|
4154
|
+
mutation: CounterfactualMutation;
|
|
4155
|
+
/** Pre-applied mutation on the step at `mutation.at`. Consumers use this
|
|
4156
|
+
* as the FIRST step the replayed agent emits (they decide whether to
|
|
4157
|
+
* re-emit it or continue from there). */
|
|
4158
|
+
mutatedStep: TrajectoryStep;
|
|
4159
|
+
}
|
|
4160
|
+
interface CounterfactualResult {
|
|
4161
|
+
counterfactualRunId: string;
|
|
4162
|
+
originalRunId: string;
|
|
4163
|
+
mutation: CounterfactualMutation;
|
|
4164
|
+
/** Structured delta summary — caller can extend via scoring. */
|
|
4165
|
+
delta: {
|
|
4166
|
+
originalOutcomeScore: number | null;
|
|
4167
|
+
counterfactualOutcomeScore: number | null;
|
|
4168
|
+
deltaScore: number | null;
|
|
4169
|
+
};
|
|
4170
|
+
}
|
|
4171
|
+
interface CounterfactualRunner {
|
|
4172
|
+
/**
|
|
4173
|
+
* Execute the agent from `ctx.prefix` with the mutation applied.
|
|
4174
|
+
* MUST emit spans into the provided emitter so they become part of
|
|
4175
|
+
* the counterfactual run. MUST call emitter.endRun() with a verdict.
|
|
4176
|
+
*/
|
|
4177
|
+
executeFrom: (ctx: CounterfactualContext, emitter: TraceEmitter) => Promise<void>;
|
|
4178
|
+
}
|
|
4179
|
+
declare function runCounterfactual(store: TraceStore, originalRunId: string, mutation: CounterfactualMutation, runner: CounterfactualRunner): Promise<CounterfactualResult>;
|
|
4180
|
+
/**
|
|
4181
|
+
* Aggregate a batch of counterfactuals into a simple attribution table:
|
|
4182
|
+
* which mutation kinds move outcomes most? (Useful when you run a grid
|
|
4183
|
+
* over the same trajectory — swap-model at every llm span, swap-tool
|
|
4184
|
+
* at every tool span — and want a ranked summary.)
|
|
4185
|
+
*/
|
|
4186
|
+
declare function attributeCounterfactuals(results: CounterfactualResult[]): Array<{
|
|
4187
|
+
mutationKind: CounterfactualMutation['kind'];
|
|
4188
|
+
n: number;
|
|
4189
|
+
meanAbsDelta: number;
|
|
4190
|
+
meanSignedDelta: number;
|
|
4191
|
+
}>;
|
|
4192
|
+
|
|
3932
4193
|
/**
|
|
3933
4194
|
* Full cross-trace diff — align two trajectories step-by-step, report
|
|
3934
4195
|
* per-step score deltas, attribute a variant's total outcome lead to
|
|
@@ -4069,6 +4330,71 @@ interface CausalAttributionReport {
|
|
|
4069
4330
|
}
|
|
4070
4331
|
declare function causalAttribution(cells: FactorialCell[]): CausalAttributionReport;
|
|
4071
4332
|
|
|
4333
|
+
/**
|
|
4334
|
+
* Process Reward Modeling — per-step rubric grading.
|
|
4335
|
+
*
|
|
4336
|
+
* A StepRubric inspects one span and returns a score + rationale.
|
|
4337
|
+
* PrmGrader applies an array of rubrics to every LLM span in a
|
|
4338
|
+
* trajectory (consumers can broaden to tool/retrieval spans via the
|
|
4339
|
+
* `kind` filter on each rubric).
|
|
4340
|
+
*
|
|
4341
|
+
* Why this matters: outcome-only eval (did the final artifact work?)
|
|
4342
|
+
* gives sparse reward — most agent turns are unattributable. PRMs
|
|
4343
|
+
* densify the signal so optimizers and RL fine-tuning can assign
|
|
4344
|
+
* credit per turn.
|
|
4345
|
+
*/
|
|
4346
|
+
|
|
4347
|
+
interface StepContext {
|
|
4348
|
+
trajectory: Trajectory;
|
|
4349
|
+
step: TrajectoryStep;
|
|
4350
|
+
/** Steps preceding `step` in trajectory order. */
|
|
4351
|
+
prior: TrajectoryStep[];
|
|
4352
|
+
/** Steps following `step`. */
|
|
4353
|
+
next: TrajectoryStep[];
|
|
4354
|
+
}
|
|
4355
|
+
interface StepRubric {
|
|
4356
|
+
id: string;
|
|
4357
|
+
/** Only grade spans of these kinds (default: all). */
|
|
4358
|
+
kinds?: Array<Span['kind']>;
|
|
4359
|
+
/** Weight in the aggregate score. Default 1. */
|
|
4360
|
+
weight?: number;
|
|
4361
|
+
/** Returns score in 0..1 + optional rationale/evidence. Return `null` to
|
|
4362
|
+
* skip grading (rubric doesn't apply to this step). */
|
|
4363
|
+
grade: (ctx: StepContext) => Promise<{
|
|
4364
|
+
score: number;
|
|
4365
|
+
rationale?: string;
|
|
4366
|
+
evidence?: string;
|
|
4367
|
+
} | null>;
|
|
4368
|
+
}
|
|
4369
|
+
interface GradedStep {
|
|
4370
|
+
spanId: string;
|
|
4371
|
+
rubricId: string;
|
|
4372
|
+
score: number;
|
|
4373
|
+
weight: number;
|
|
4374
|
+
rationale?: string;
|
|
4375
|
+
evidence?: string;
|
|
4376
|
+
}
|
|
4377
|
+
interface PrmGradedTrace {
|
|
4378
|
+
runId: string;
|
|
4379
|
+
steps: GradedStep[];
|
|
4380
|
+
/** Weighted mean of all graded steps; 0..1. */
|
|
4381
|
+
aggregateScore: number;
|
|
4382
|
+
/** Number of spans graded — useful for sanity-checking coverage. */
|
|
4383
|
+
gradedCount: number;
|
|
4384
|
+
/** Number of spans in the trajectory that no rubric matched. */
|
|
4385
|
+
ungradedCount: number;
|
|
4386
|
+
}
|
|
4387
|
+
declare class PrmGrader {
|
|
4388
|
+
private rubrics;
|
|
4389
|
+
constructor(rubrics: StepRubric[]);
|
|
4390
|
+
/**
|
|
4391
|
+
* Grade every eligible span in a run. Emits a JudgeVerdict span for each
|
|
4392
|
+
* (rubric × span) verdict so the result is visible to downstream pipelines
|
|
4393
|
+
* (judgeAgreementView, etc.) — PRM is just "a judge that runs per span."
|
|
4394
|
+
*/
|
|
4395
|
+
grade(store: TraceStore, runId: string): Promise<PrmGradedTrace>;
|
|
4396
|
+
}
|
|
4397
|
+
|
|
4072
4398
|
/**
|
|
4073
4399
|
* Reward-model export — the productizable wrapper around PRM training
|
|
4074
4400
|
* data. Takes a TraceStore + PrmGrader, produces an embeddable
|
|
@@ -5410,6 +5736,23 @@ declare function precision<T>(goldens: GoldenSpec[], candidates: T[], options?:
|
|
|
5410
5736
|
text?: (candidate: T) => string;
|
|
5411
5737
|
}): number;
|
|
5412
5738
|
|
|
5739
|
+
/**
|
|
5740
|
+
* LockedJsonlAppender — mutex-serialized JSONL append helper for arbitrary
|
|
5741
|
+
* payloads. The reference-replay store does the same thing for typed
|
|
5742
|
+
* `ReferenceReplayRun` rows; this is the generic version used by
|
|
5743
|
+
* `MutationTelemetry`, `TrialTelemetry`, and any other consumer that wants
|
|
5744
|
+
* append-only durable telemetry without rolling its own lock.
|
|
5745
|
+
*
|
|
5746
|
+
* Locks are per absolute file path (process-local). Cross-process
|
|
5747
|
+
* concurrency is NOT addressed — that's an fcntl/flock problem.
|
|
5748
|
+
*/
|
|
5749
|
+
declare class LockedJsonlAppender {
|
|
5750
|
+
readonly path: string;
|
|
5751
|
+
private readonly mutex;
|
|
5752
|
+
constructor(path: string);
|
|
5753
|
+
append(entry: unknown): Promise<void>;
|
|
5754
|
+
}
|
|
5755
|
+
|
|
5413
5756
|
/**
|
|
5414
5757
|
* Inter-critic / inter-pass orthogonality.
|
|
5415
5758
|
*
|
|
@@ -6256,6 +6599,247 @@ declare function attest(report: unknown, provenance: AttestationProvenance): Att
|
|
|
6256
6599
|
*/
|
|
6257
6600
|
declare function verifyAttestation(report: unknown, attested: AttestedReport): AttestationVerification;
|
|
6258
6601
|
|
|
6602
|
+
/**
|
|
6603
|
+
* Export side of the product benchmark bundle contract: convert product
|
|
6604
|
+
* eval run directories (`records.jsonl` of `RunRecord` rows + trace/raw
|
|
6605
|
+
* artifacts) into a portable `product-benchmark-manifest.json` +
|
|
6606
|
+
* `product-benchmark-records.jsonl` bundle that
|
|
6607
|
+
* `validateProductBenchmarkRun` accepts.
|
|
6608
|
+
*
|
|
6609
|
+
* Product-specific policy (safety-split detection, tool-call recovery,
|
|
6610
|
+
* profile id fallback, artifact materialization) enters through explicit
|
|
6611
|
+
* options; everything else is the shared union of the tax/legal/creative
|
|
6612
|
+
* exporters. Scenario catalogs, smoke runners, and CLIs stay in the
|
|
6613
|
+
* products.
|
|
6614
|
+
*
|
|
6615
|
+
* Input rows are checked structurally, not with `validateRunRecord`:
|
|
6616
|
+
* product harnesses record bare model aliases and partial provenance, and
|
|
6617
|
+
* the bundle contract's own validators re-check every field that matters
|
|
6618
|
+
* on the way out.
|
|
6619
|
+
*/
|
|
6620
|
+
|
|
6621
|
+
/** Full mutable-surface superset a product arm may declare. */
|
|
6622
|
+
declare const productBenchmarkMutableSurfaces: readonly ["prompt", "resources.files", "tools", "mcp", "hooks", "subagents"];
|
|
6623
|
+
interface ProductBenchmarkExportOptions {
|
|
6624
|
+
/** Source eval run directories, each containing a `records.jsonl` of RunRecord rows. */
|
|
6625
|
+
readonly runDirs: readonly string[];
|
|
6626
|
+
/** Destination directory for the bundle (manifest + records + materialized source runs). */
|
|
6627
|
+
readonly outDir: string;
|
|
6628
|
+
readonly projectId: string;
|
|
6629
|
+
readonly benchmarkId: string;
|
|
6630
|
+
/** Repo-relative path of the product's canonical agent profile source. */
|
|
6631
|
+
readonly agentProfilePath: string;
|
|
6632
|
+
/** Pass threshold applied when a row carries no explicit `outcome.raw.pass`. Default 0.7. */
|
|
6633
|
+
readonly passThreshold?: number;
|
|
6634
|
+
/**
|
|
6635
|
+
* First scenario tag. Defaults to `projectId` with a trailing `-agent`
|
|
6636
|
+
* stripped (`tax-agent` → `tax`), matching the product exporters.
|
|
6637
|
+
*/
|
|
6638
|
+
readonly scenarioTagPrefix?: string;
|
|
6639
|
+
/** Profile id used when a row has no `agentProfile.profileId`. Defaults to the row's arm id. */
|
|
6640
|
+
readonly fallbackProfileId?: string;
|
|
6641
|
+
/** Arm mutable surfaces recorded in the manifest. Defaults to the full superset. */
|
|
6642
|
+
readonly mutableSurfaces?: readonly string[];
|
|
6643
|
+
/**
|
|
6644
|
+
* Copy each run dir into `<outDir>/source-runs/` and record
|
|
6645
|
+
* bundle-relative artifact paths (portable, self-contained). When false,
|
|
6646
|
+
* artifacts keep absolute paths into the original run dirs. Default true.
|
|
6647
|
+
*/
|
|
6648
|
+
readonly materializeSourceRuns?: boolean;
|
|
6649
|
+
/**
|
|
6650
|
+
* Override split classification for a row. Return undefined to fall back
|
|
6651
|
+
* to the default (`outcome.raw.safety === 1` → safety, then splitTag).
|
|
6652
|
+
*/
|
|
6653
|
+
readonly classifySplit?: (record: RunRecord) => ProductBenchmarkSplit | undefined;
|
|
6654
|
+
/** Recovers a tool-call count when the row's raw bag carries none (e.g. from turn artifacts). */
|
|
6655
|
+
readonly toolCallFallback?: (record: RunRecord, runDir: string) => number;
|
|
6656
|
+
/** Backend version recorded per row. Defaults to the cwd package.json's `@tangle-network/sandbox` range. */
|
|
6657
|
+
readonly backendVersion?: string;
|
|
6658
|
+
/**
|
|
6659
|
+
* Explicit substrate versions for the manifest, merged over what the cwd
|
|
6660
|
+
* package.json / node_modules resolve. Use when a substrate package is not
|
|
6661
|
+
* installed where the export runs — the validator refuses an `'unknown'`
|
|
6662
|
+
* version, so provide the real one rather than shipping the sentinel.
|
|
6663
|
+
*/
|
|
6664
|
+
readonly substrate?: Partial<ProductBenchmarkManifest['substrate']>;
|
|
6665
|
+
}
|
|
6666
|
+
interface ProductBenchmarkSingleRunExportOptions extends Omit<ProductBenchmarkExportOptions, 'runDirs'> {
|
|
6667
|
+
readonly runDir: string;
|
|
6668
|
+
}
|
|
6669
|
+
interface ProductBenchmarkExportResult {
|
|
6670
|
+
readonly manifestPath: string;
|
|
6671
|
+
readonly recordsPath: string;
|
|
6672
|
+
readonly records: number;
|
|
6673
|
+
}
|
|
6674
|
+
/** Repo identity from the exporting process's cwd. `'unknown'` values are flagged by `validateProductBenchmarkRun`. */
|
|
6675
|
+
declare function productBenchmarkRepoIdentity(): ProductBenchmarkManifest['repo'];
|
|
6676
|
+
/** Map one RunRecord row to a validated product benchmark record. */
|
|
6677
|
+
declare function runRecordToProductBenchmarkRecord(record: RunRecord, runDir: string, artifactRoot: string, artifacts: ProductBenchmarkRecord['artifacts'], options: ProductBenchmarkExportOptions | ProductBenchmarkSingleRunExportOptions): ProductBenchmarkRecord;
|
|
6678
|
+
/** Derive the bundle manifest from already-normalized records. */
|
|
6679
|
+
declare function buildProductBenchmarkManifest(records: readonly ProductBenchmarkRecord[], options: Pick<ProductBenchmarkExportOptions, 'outDir' | 'projectId' | 'benchmarkId' | 'scenarioTagPrefix' | 'mutableSurfaces' | 'substrate'>): ProductBenchmarkManifest;
|
|
6680
|
+
/** Single-run convenience wrapper over `exportProductBenchmarkRuns`. */
|
|
6681
|
+
declare function exportProductBenchmark(options: ProductBenchmarkSingleRunExportOptions): ProductBenchmarkExportResult;
|
|
6682
|
+
/**
|
|
6683
|
+
* Export one or more product eval run dirs into a validated product
|
|
6684
|
+
* benchmark bundle at `outDir`. Both the manifest and every record are
|
|
6685
|
+
* run through the contract validators before anything is written.
|
|
6686
|
+
*/
|
|
6687
|
+
declare function exportProductBenchmarkRuns(options: ProductBenchmarkExportOptions): ProductBenchmarkExportResult;
|
|
6688
|
+
|
|
6689
|
+
declare const productBenchmarkSplits: readonly ["practice", "dev", "holdout", "safety", "sentinel"];
|
|
6690
|
+
type ProductBenchmarkSplit = (typeof productBenchmarkSplits)[number];
|
|
6691
|
+
interface ProductBenchmarkRepoRef {
|
|
6692
|
+
readonly url: string;
|
|
6693
|
+
readonly commit: string;
|
|
6694
|
+
readonly branch: string;
|
|
6695
|
+
}
|
|
6696
|
+
interface ProductBenchmarkSubstrateVersions {
|
|
6697
|
+
readonly agentEval: string;
|
|
6698
|
+
readonly agentRuntime: string;
|
|
6699
|
+
readonly agentInterface: string;
|
|
6700
|
+
readonly sandbox: string;
|
|
6701
|
+
readonly agentBench?: string;
|
|
6702
|
+
}
|
|
6703
|
+
interface ProductBenchmarkProfileRef {
|
|
6704
|
+
readonly id: string;
|
|
6705
|
+
readonly profileHash: string;
|
|
6706
|
+
readonly agentProfilePath: string;
|
|
6707
|
+
}
|
|
6708
|
+
interface ProductBenchmarkArm {
|
|
6709
|
+
readonly id: string;
|
|
6710
|
+
readonly profileId: string;
|
|
6711
|
+
readonly mutableSurfaces: readonly string[];
|
|
6712
|
+
readonly policyAxes: Record<string, unknown>;
|
|
6713
|
+
}
|
|
6714
|
+
interface ProductBenchmarkScenario {
|
|
6715
|
+
readonly id: string;
|
|
6716
|
+
readonly split: ProductBenchmarkSplit;
|
|
6717
|
+
readonly tags: readonly string[];
|
|
6718
|
+
readonly sourceAllowedForSynthesis: boolean;
|
|
6719
|
+
}
|
|
6720
|
+
interface ProductBenchmarkBudgets {
|
|
6721
|
+
readonly maxUsd: number;
|
|
6722
|
+
readonly maxCells: number;
|
|
6723
|
+
readonly maxWallMs: number;
|
|
6724
|
+
}
|
|
6725
|
+
interface ProductBenchmarkManifest {
|
|
6726
|
+
readonly schemaVersion: 1;
|
|
6727
|
+
readonly projectId: string;
|
|
6728
|
+
readonly benchmarkId: string;
|
|
6729
|
+
readonly repo: ProductBenchmarkRepoRef;
|
|
6730
|
+
readonly substrate: ProductBenchmarkSubstrateVersions;
|
|
6731
|
+
readonly profiles: readonly ProductBenchmarkProfileRef[];
|
|
6732
|
+
readonly arms: readonly ProductBenchmarkArm[];
|
|
6733
|
+
readonly scenarios: readonly ProductBenchmarkScenario[];
|
|
6734
|
+
readonly budgets: ProductBenchmarkBudgets;
|
|
6735
|
+
readonly expectedArtifactDir: string;
|
|
6736
|
+
}
|
|
6737
|
+
interface AgentProfileRuntimeReceipt {
|
|
6738
|
+
readonly model: string;
|
|
6739
|
+
readonly harness: string;
|
|
6740
|
+
readonly backend: string;
|
|
6741
|
+
readonly reasoningEffort?: string;
|
|
6742
|
+
}
|
|
6743
|
+
type RuntimeResolution = AgentProfileRuntimeReceipt;
|
|
6744
|
+
interface ProductBenchmarkRecord {
|
|
6745
|
+
readonly schemaVersion: 1;
|
|
6746
|
+
readonly projectId: string;
|
|
6747
|
+
readonly benchmarkId: string;
|
|
6748
|
+
readonly runId: string;
|
|
6749
|
+
readonly scenarioId: string;
|
|
6750
|
+
readonly split: ProductBenchmarkSplit;
|
|
6751
|
+
readonly armId: string;
|
|
6752
|
+
readonly rep: number;
|
|
6753
|
+
readonly agentProfile: {
|
|
6754
|
+
readonly id: string;
|
|
6755
|
+
readonly hash: string;
|
|
6756
|
+
readonly path: string;
|
|
6757
|
+
readonly declared: RuntimeResolution;
|
|
6758
|
+
readonly resolved: RuntimeResolution;
|
|
6759
|
+
};
|
|
6760
|
+
readonly model: {
|
|
6761
|
+
readonly provider: string;
|
|
6762
|
+
readonly id: string;
|
|
6763
|
+
};
|
|
6764
|
+
readonly backend: {
|
|
6765
|
+
readonly kind: string;
|
|
6766
|
+
readonly version: string;
|
|
6767
|
+
};
|
|
6768
|
+
readonly outcome: {
|
|
6769
|
+
readonly pass: boolean;
|
|
6770
|
+
readonly score: number;
|
|
6771
|
+
readonly dimensions: Record<string, number>;
|
|
6772
|
+
readonly failureMode: string | null;
|
|
6773
|
+
};
|
|
6774
|
+
readonly usage: {
|
|
6775
|
+
readonly inputTokens: number;
|
|
6776
|
+
readonly outputTokens: number;
|
|
6777
|
+
readonly costUsd: number;
|
|
6778
|
+
readonly wallMs: number;
|
|
6779
|
+
readonly toolCalls: number;
|
|
6780
|
+
};
|
|
6781
|
+
readonly integrity: {
|
|
6782
|
+
readonly realBackend: boolean;
|
|
6783
|
+
readonly rawCapture: boolean;
|
|
6784
|
+
readonly traceCapture: boolean;
|
|
6785
|
+
readonly noStubRows: boolean;
|
|
6786
|
+
readonly priced: boolean;
|
|
6787
|
+
readonly profileMaterialized: boolean;
|
|
6788
|
+
};
|
|
6789
|
+
readonly artifacts: {
|
|
6790
|
+
readonly records: string;
|
|
6791
|
+
readonly traces: string;
|
|
6792
|
+
readonly raws: string;
|
|
6793
|
+
readonly scores: string;
|
|
6794
|
+
readonly workspace: string;
|
|
6795
|
+
};
|
|
6796
|
+
}
|
|
6797
|
+
interface ProductBenchmarkRunInput {
|
|
6798
|
+
readonly manifestPath: string;
|
|
6799
|
+
readonly recordsPath: string;
|
|
6800
|
+
readonly artifactRoot?: string;
|
|
6801
|
+
readonly checkArtifacts?: boolean;
|
|
6802
|
+
}
|
|
6803
|
+
interface ProductBenchmarkValidationReport {
|
|
6804
|
+
readonly manifestPath: string;
|
|
6805
|
+
readonly recordsPath: string;
|
|
6806
|
+
readonly records: number;
|
|
6807
|
+
/** Manifest repo fields that are empty or the `'unknown'` export sentinel. */
|
|
6808
|
+
readonly repoFailures: readonly string[];
|
|
6809
|
+
/** Manifest substrate versions that are empty or the `'unknown'` export
|
|
6810
|
+
* sentinel — a bundle without substrate identity is not reproducible. */
|
|
6811
|
+
readonly substrateFailures: readonly string[];
|
|
6812
|
+
readonly projects: readonly string[];
|
|
6813
|
+
readonly benchmarks: readonly string[];
|
|
6814
|
+
readonly arms: readonly string[];
|
|
6815
|
+
readonly scenarios: readonly string[];
|
|
6816
|
+
readonly passed: number;
|
|
6817
|
+
readonly failed: number;
|
|
6818
|
+
readonly inputTokens: number;
|
|
6819
|
+
readonly outputTokens: number;
|
|
6820
|
+
readonly costUsd: number;
|
|
6821
|
+
readonly wallMs: number;
|
|
6822
|
+
readonly integrityFailures: readonly string[];
|
|
6823
|
+
readonly missingArtifacts: readonly string[];
|
|
6824
|
+
}
|
|
6825
|
+
interface ProductBenchmarkArtifactPaths {
|
|
6826
|
+
readonly manifestPath: string;
|
|
6827
|
+
readonly recordsPath: string;
|
|
6828
|
+
}
|
|
6829
|
+
declare function validateProductBenchmarkManifest(value: unknown): ProductBenchmarkManifest;
|
|
6830
|
+
declare function validateProductBenchmarkRecord(value: unknown): ProductBenchmarkRecord;
|
|
6831
|
+
declare function productBenchmarkIntegrityFailures(record: ProductBenchmarkRecord): string[];
|
|
6832
|
+
declare function readProductBenchmarkRecords(path: string): ProductBenchmarkRecord[];
|
|
6833
|
+
declare function readProductBenchmarkManifest(path: string): ProductBenchmarkManifest;
|
|
6834
|
+
declare function validateProductBenchmarkRun(input: ProductBenchmarkRunInput): ProductBenchmarkValidationReport;
|
|
6835
|
+
declare function findProductBenchmarkArtifacts(runDir: string): ProductBenchmarkArtifactPaths | null;
|
|
6836
|
+
/**
|
|
6837
|
+
* Fail-loud gate over a bundle directory: locates the manifest + records,
|
|
6838
|
+
* runs `validateProductBenchmarkRun`, and throws with every repo,
|
|
6839
|
+
* integrity, and artifact failure listed. Returns the report when clean.
|
|
6840
|
+
*/
|
|
6841
|
+
declare function assertProductBenchmarkRun(runDir: string): ProductBenchmarkValidationReport;
|
|
6842
|
+
|
|
6259
6843
|
/**
|
|
6260
6844
|
* Content-addressed judge-verdict caching.
|
|
6261
6845
|
*
|
|
@@ -6334,4 +6918,4 @@ type CachedJudge<TArtifact, TScenario extends Scenario$1 = Scenario$1> = JudgeCo
|
|
|
6334
6918
|
*/
|
|
6335
6919
|
declare function cachedJudge<TArtifact, TScenario extends Scenario$1 = Scenario$1>(judge: JudgeConfig<TArtifact, TScenario>, store: VerdictCacheStore, options: CachedJudgeOptions): CachedJudge<TArtifact, TScenario>;
|
|
6336
6920
|
|
|
6337
|
-
export { ATTESTATION_ALGORITHM, type ActiveLearningOptions, type AdapterRun, AgentDriver, type AgentDriverConfig, AgentEvalError, type AgreementResult, type AlignmentOp, AnalyzeTracesInput, AnalyzeTracesOptions, AnalyzeTracesResult, type AntiSlopConfig, type AntiSlopIssue, type AntiSlopReport, type AssertCapabilityHeadroomOptions, type AssertCrossFamilyOptions, type AssertSingleBackendOptions, type AttestationProvenance, type AttestationVerification, type AttestedReport, type AutoPrClient, AxGepaSteeringOptimizer, type AxSteeringOptimizerConfig, type BackendDescriptor, BaselineReport, BehaviorAssertion, BehavioralMetrics, BenchmarkReport, BenchmarkRunner, BenchmarkRunnerConfig, type BisectOptions, type BisectResult, type BisectStep, type BlendWeights, BudgetBreachError, BudgetGuard, BudgetLedgerEntry, BudgetSpec, type BuildAgreementJudgeOptions, CODING_HARNESSES, type CachedJudge, type CachedJudgeOptions, CallExpectation, type CanaryAlert, type CanaryKind, type CanaryLeak, type CanaryOptions, type CanaryReport, type CanarySeverity, type CandidateComparison, type CandidateScenario, type CapabilityHeadroomOptions, type CapabilityHeadroomResult, type CausalAttributionReport, type CellVerdict, ChannelRollup, ChatRequest, CheckResult, CollectedArtifacts, type CommandRunner, type CompareLabels, type ComparePairedArmsOptions, CompletionCriterion, ConfigError, type ContinuityCheck, type ContinuityCheckResult, type ContinuityReport, type ContinuitySnapshotPair, type ContractCheckResult, type ContractJudgeOptions, type ContractMetric, type ContractReport, type ContractRule, type ContractRuleKind, type ContractSpan, type ContractVerdict, type ContractViolation, ConvergenceTracker, CorrectnessChecker, type CostEntry, CostLedger, type CostReport, type CostSummary, CostTracker, CreateChatClientOpts, type CreateDefaultReviewerOptions, type CreateExperimentInput, type CreateSandboxPoolOpts, CrossFamilyError, type CrossTraceDiff, type CrossTraceDiffOptions, DEFAULT_AGENT_SLOS, DEFAULT_FINDERS, DEFAULT_MUTATION_PRIMITIVES, DEFAULT_MUTATORS, DEFAULT_PR_REVIEW_SCORE_WEIGHTS, DEFAULT_SEVERITY_WEIGHTS, Dataset, DatasetScenario, type DecideNextUserTurnOpts, DefaultVerdict, type DeployFamily, type DeployGateLayerInput, type DeployRunResult, type DeployRunner, type DescriptionLengthCandidate, type DescriptionLengthConfig, type DescriptionLengthDecision, type DescriptionLengthEvidence, DescriptionLengthGate, type DescriptionLengthRejectionCode, type DetectorEvent, type DetectorSeverity, type DetectorSignal, type DiffScorecardOptions, type DirEntry, type DiscoverPersonasOptions, type DiscoveredPersona, DriverResult, DriverState, DualAgentBench, type DualAgentBenchConfig, type DualAgentReport, type DualAgentRound, type DualAgentScenario, type DualAgentScenarioResult, ERROR_COUNT_PATTERNS, type EnsembleAggregate, type EnsembleJudgeOptions, type ErrorCountPattern, type ErrorStreakOptions, type EvalToolDef, EvalTraceStore, type EvolutionRound, type ExecutorConfig, type Expectation, type Experiment, type ExperimentProvenance, type ExperimentRep, type ExperimentStats, type ExperimentStore, ExperimentTracker, type ExperimentTrackerOptions, type ExperimentVerdict, type ExportedRewardModel, type ExtractOptions, type ExtractResult, type FactorContribution, type FactorialCell, FailureClass, FeedbackLabel, FeedbackTrajectory, FeedbackTrajectoryStore, type FieldAgreementSpec, type FieldDestination, type FileChange, type FlowAction, type FlowLayerEnv, type FlowLayerFactoryInput, type FlowRunner, type FlowRunnerStepResult, type FlowSpec, type FlowStep, type GhCliClientOptions, type GoldScenario, type GoldSplit, type GoldenSeverity, type GoldenSpec, HARNESS_NATIVE_MODEL, HarnessConfig, type HeadroomClass, type HeadroomInput, type HeldOutPartition, type HiddenCriteriaGrader, type HiddenGradeResult, type HiddenLeak, HoldoutAuditor, type HttpGithubClientOptions, INTENT_MATCH_JUDGE_VERSION, type ImageData, type ImprovementThresholds, type ImprovementVerdictResult, InMemoryWorkspaceInspector, type InferenceScorer, type InspectorContext, type IntentMatchInput, type IntentMatchOptions, type IntentMatchResult, type InteractionContribution, JudgeError, type JudgeFamily, type JudgeFleetOptions, JudgeFn, JudgeParseError, type JudgeReplayResult, type JudgeRetryOutcome, type JudgeRetryPolicy, JudgeRunner, type JudgeScoreInput, type JudgeVerdict, type KeywordConceptSpec, type KeywordCoverageFinding, type KeywordCoverageOptions, type KeywordCoverageResult, type LangfuseEnvelope, type LangfuseGeneration, type LangfuseScore, Layer, LayerResult, type LeaderboardOptions, type LeaderboardRow, type LiveProofArtifact, type LiveProofConfig, type LiveProofContext, type LiveProofResult, LlmClientOptions, LlmSpan, MODEL_PRICING, type MakeEvalToolsConfig, type MatchResult, type MatchedPair, type MatcherResult, McNemarResult, type MergeOptions, MetricsCollector, type ModelCostRollup, type ModelPreflight, type ModelSeats, ModelsUnreachableError, type MuffledFinder, type MuffledFinding, type MultiToolchainLayerConfig, type Mutator, Mutex, type NoLeakOptions, type NoProgressOptions, type Oracle, type OracleObservation, type OracleReport, type OracleResult, type OrthogonalityInput, type OrthogonalityResult, OtelExportConfig, OtelExporter, type OtelPipelineHandle, type OtelPipelineOptions, type PairArmsOptions, type PairArmsResult, type PairedArmRow, type PairedArmsComparison, PairedBootstrapOptions, PairedBootstrapResult, type PairedCorrectness, type PairedMetricDelta, PairwiseSteeringOptimizer, type ParaphraseRobustnessScenarioInput, type ParaphraseRobustnessScenarioResult, type ParseStudentLabel, type PartitionHeldOutOptions, PersonaConfig, type Playbook, type PlaybookEntry, type PoolSlot, type PrReviewAuditCase, type PrReviewBenchmarkSummary, type PrReviewComment, type PrReviewMatchedFinding, type PrReviewOutcome, type PrReviewReferenceFinding, type PrReviewScore, type PrReviewScoreWeights, type PrReviewSeverity, type PrReviewSource, type PreflightModelsOptions, type PreflightOutcome, ProductClient, ProductClientConfig, type ProfileAxisSpec, type PromptHandle, PromptRegistry, type ProposeAutomatedPullRequestInput, type ProposeAutomatedPullRequestResult, type ProvenanceReader, type RecordRunsOptions, type ReferenceMatchResult, type ReferenceReplayAdapter, type ReferenceReplayAdapterFn, type ReferenceReplayAdapterLike, type ReferenceReplayAggregate, type ReferenceReplayCandidate, type ReferenceReplayCase, type ReferenceReplayCaseRun, type ReferenceReplayExecutionScenario, type ReferenceReplayItem, type ReferenceReplayMatch, type ReferenceReplayMatchStrategy, type ReferenceReplayMatcher, type ReferenceReplayPromotionDecision, type ReferenceReplayPromotionPolicy, type ReferenceReplayRun, type ReferenceReplayRunContext, type ReferenceReplayRunOptions, type ReferenceReplayRunStore, type ReferenceReplayScenario, type ReferenceReplayScenarioScore, type ReferenceReplayScore, type ReferenceReplayScoreOptions, type ReferenceReplaySplit, type ReferenceReplaySplitComparison, type ReferenceReplaySteeringRowsOptions, type ReflectionContext, type ReflectionProposal, ReleaseConfidenceScorecard, ReleaseConfidenceThresholds, type RenderStudentPrompt, type RepeatedActionOptions, type RepoRef, type ReviewerMemoryEntry, type ReviewerOutput, type ReviewerPromptInput, type ReviewerSoftFailDefaults, type ReviewerVerificationSummary, RiskDifferenceResult, type RobustnessResult, type RoutedField, Run, type RunCommandInput, type RunCommandResult, type RunDistillationOptions, type RunDistillationResult, RunFilter, RunRecord, type RunRecordBackend, type RunRecordFilter, RunScore, RunScoreWeights, RunSplitTag, SandboxDriver, SandboxHarnessResult, type SandboxJudgeKind, type SandboxJudgeResult, type SandboxJudgeSpec, type SandboxPool, type ScanOptions, Scenario, type ScenarioCost, ScenarioFile, ScenarioRegistry, ScenarioResult, type Scorecard, type ScorecardCell, type ScorecardCellDiff, type ScorecardDiff, type ScorecardEntry, type ScorecardLogLine, type ScoredTarget, type SeatName, type SeatPresetName, SeatUnsetError, type SelfPlayOptions, type SelfPlayProposer, type SelfPlayScorer, type SerializedRegex, Severity, type SingleBackendDivergence, SingleBackendError, type SingleBackendField, type SingleBackendReport, type Slo, type SloCheckResult, type SloComparator, type SloReport, type SloSeverity, type SlopCategory, type SlotFactory, type SpanPredicate, type SplitGoldOptions, SteeringBundle, type SteeringOptimizationResult, type SteeringOptimizationRow, type SteeringOptimizationSelector, type SteeringOptimizerBackend, type SteeringOptimizerConfig, type StepAttribution, type StreamingDetector, type SynthesisReason, type SynthesisTarget, type TaskHeadroom, TestResult, type TextMatcher, type ThresholdContract, TokenCounter, type TokenSpec, type ToolMatcher, ToolSpan, TraceAnalystSpan, type TraceContract, TraceContractBuilder, TraceEmitter, TraceStore, type TracedAnalystOptions, type TracedJudgeOptions, Trajectory, TrajectoryStep, type TreatmentClass, type TreatmentGate, type TreatmentGateInput, type TreatmentGateOptions, type TrialTrace, TurnMetrics, UI_FINDING_SEVERITIES, UI_LENSES, UNIVERSAL_FINDERS, type UiFinding, type UiFindingScreenshot, type UiFindingSeverity, type UiLens, type VerdictCacheStats, type VerdictCacheStore, VerifyContext, type VisualDiffOptions, type VisualDiffResult, type ViteDeployRunnerInput, type WorkerDriverContext, type WorkspaceAssertion, type WorkspaceAssertionResult, type WorkspaceInspector, type WorkspaceSnapshot, type WranglerDeployRunnerInput, adversarialJudge, agentProfileHash, agentProfileId, agentProfileModelId, agentVisibleFields, aggregateJudgeVerdicts, aggregatePrReviewScore, analyzeAntiSlop, appendScorecard, assertCapabilityHeadroom, assertCrossFamily, assertModelsServed, assertNoHiddenLeak, assertSingleBackend, assignHeldOutTag, attachCostToReport, attest, bisect, blendHeldout, buildAgreementJudge, buildDriverSystemPrompt, buildReflectionPrompt, buildReviewerPrompt, buildWorkerDriverSystemPrompt, cachedJudge, canaryLeakView, canonicalJson, capabilityHeadroom, causalAttribution, checkBehavioralCanary, checkCanaries, checkSlos, checkTraceContracts, classifyTreatment, codeExecutionJudge, coherenceJudge, collectionPreserved, commentsForSource, commitBisect, comparePairedArms, compareReferenceReplay, compilerJudge, computeExperimentStats, contentHash, contractJudge, costReport, createAntiSlopJudge, createCustomJudge, createDefaultReviewer, createDomainExpertJudge, createIntentMatchJudge, createSandboxPool, crossTraceDiff, dataDescriptionBits, decideNextUserTurn, decideReferenceReplayPromotion, decideReferenceReplayRunPromotion, defaultBlendWeights, defaultJudges, defaultParseStudentLabel, defaultReferenceReplayMatcher, defaultRenderStudentPrompt, deployGateLayer, diffScorecard, discoverPersonas, distillPlaybook, ensembleJudge, errorStreakDetector, estimateCost, estimateTokens, evaluateContract, evaluateOracles, evaluateTraceContract, executeScenario, expandProfileAxes, expectAgent, exportRewardModel, extractAssetUrls, extractErrorCount, fieldAgreement, fileContains, fileExists, fileExperimentStore, fileVerdictCache, findAutoMatchNoExpectation, findConstructorCwdDropped, findFallbackToPass, findLiteralTruePass, findSkipCountsAsPass, flowLayer, fnv1a32, formatBenchmarkReport, formatDriverReport, formatFindings, formatScorecardDiff, gateTreatmentApplied, gateTreatmentFromMetrics, gateTreatmentFromSpans, gateTreatmentFromToolSpans, ghCliClient, gitProvenanceReader, precision as goldenPrecision, gradeOnHidden, harnessAxisOf, hashContent, hashToUnit, hiddenGrade, htmlContainsElement, httpGithubClient, improvementVerdict, inMemoryExperimentStore, inMemoryReferenceReplayStore, inMemoryRunRecordBackend, inMemoryVerdictCache, isHiddenDestination, isModelPriced, isOtelConfigured, jsonShape, jsonlReferenceReplayStore, jsonlRunRecordBackend, judgeFamily, keyPreserved, leaderboard, linterJudge, loadGoldScenarios, loadScorecard, loadScorerFromGrader, localCommandRunner, lowercaseMutator, makeEvalTools, matchGoldens, matchSpan, mergeLayerResults, modelDescriptionBits, multiToolchainLayer, noProgressDetector, notBlocked, observeAll, pairArms, paraphraseRobustness, paraphraseRobustnessScenarios, parseGoldJsonl, parseReflectionResponse, partitionHeldOut, passOrthogonality, pixelDeltaRatio, politenessPrefixMutator, preflightModels, printDriverSummary, index as profile, promptBisect, proposeSynthesisTargets, recordRuns, recordRunsToScorecard, referenceReplayRunsToSteeringRows, referenceReplayScenarioToRunScore, regexMatches, renderMarkdownReport, renderPlaybookMarkdown, repeatedActionDetector, replayScorerOverCorpus, replayTraceThroughJudge, resolveModelPricing, resolveSeat, routeFields, rowCount, rowWhere, runAssertions, runBehavioralCanaries, runCanaries, runDistillation, runE2EWorkflow, runExpectations, runIntentMatchJudge, runJudgeFleet, runKeywordCoverageJudge, runKeywordCoverageJudgeUrl, runLiveProof, runReferenceReplay, runScore, runSelfPlay, scanForMuffledGates, scoreContinuity, scorePrReviewComments, scorePrReviewSource, scoreReferenceReplay, seatPresets, securityJudge, sentenceReorderMutator, splitGold, statusAdvanced, summarizePrReviewBenchmark, testJudge, textInSnapshot, toLangfuseEnvelope, toOpenAiTool, toPrometheusText, traceContract, traceJudge, traceJudgeEnsemble, tracedAnalyzeTraces, typoMutator, urlContains, verifyAttestation, visualDiff, viteDeployRunner, weightedRecall, whitespaceCollapseMutator, withHeldoutBlend, withJudgeRetry, withOtelPipeline, wranglerDeployRunner };
|
|
6921
|
+
export { ATTESTATION_ALGORITHM, type ActiveLearningOptions, type AdapterRun, AgentDriver, type AgentDriverConfig, AgentEvalError, type AgentProfileRuntimeReceipt, type AgreementResult, type AlignmentOp, AnalyzeTracesInput, AnalyzeTracesOptions, AnalyzeTracesResult, type AntiSlopConfig, type AntiSlopIssue, type AntiSlopReport, type AssertCapabilityHeadroomOptions, type AssertCrossFamilyOptions, type AssertSingleBackendOptions, type AttestationProvenance, type AttestationVerification, type AttestedReport, type AutoPrClient, AxGepaSteeringOptimizer, type AxSteeringOptimizerConfig, type BackendDescriptor, BaselineReport, BehaviorAssertion, BehavioralMetrics, BenchmarkReport, BenchmarkRunner, BenchmarkRunnerConfig, type BisectOptions, type BisectResult, type BisectStep, type BlendWeights, BudgetBreachError, BudgetGuard, BudgetLedgerEntry, BudgetSpec, type BuildAgreementJudgeOptions, CODING_HARNESSES, type CachedJudge, type CachedJudgeOptions, CallExpectation, type CanaryAlert, type CanaryKind, type CanaryLeak, type CanaryOptions, type CanaryReport, type CanarySeverity, type CandidateComparison, type CandidateScenario, type CapabilityHeadroomOptions, type CapabilityHeadroomResult, type CausalAttributionReport, type CellVerdict, ChannelRollup, ChatRequest, CheckResult, CollectedArtifacts, type CommandRunner, type CompareLabels, type ComparePairedArmsOptions, CompletionCriterion, ConfigError, type ContinuityCheck, type ContinuityCheckResult, type ContinuityReport, type ContinuitySnapshotPair, type ContractCheckResult, type ContractJudgeOptions, type ContractMetric, type ContractReport, type ContractRule, type ContractRuleKind, type ContractSpan, type ContractVerdict, type ContractViolation, ControlEvalResult, ControlSeverity, ConvergenceTracker, CorrectnessChecker, type CostEntry, CostLedger, type CostReport, type CostSummary, CostTracker, type CounterfactualContext, type CounterfactualMutation, type CounterfactualResult, type CounterfactualRunner, CreateChatClientOpts, type CreateDefaultReviewerOptions, type CreateExperimentInput, type CreateSandboxPoolOpts, CrossFamilyError, type CrossTraceDiff, type CrossTraceDiffOptions, DEFAULT_AGENT_SLOS, DEFAULT_FINDERS, DEFAULT_HARNESS_OBJECTIVES, DEFAULT_MUTATION_PRIMITIVES, DEFAULT_MUTATORS, DEFAULT_PR_REVIEW_SCORE_WEIGHTS, DEFAULT_SEVERITY_WEIGHTS, type DataAcquisitionPlan, Dataset, DatasetScenario, type DecideNextUserTurnOpts, DefaultVerdict, type DeployFamily, type DeployGateLayerInput, type DeployRunResult, type DeployRunner, type DescriptionLengthCandidate, type DescriptionLengthConfig, type DescriptionLengthDecision, type DescriptionLengthEvidence, DescriptionLengthGate, type DescriptionLengthRejectionCode, type DetectorEvent, type DetectorSeverity, type DetectorSignal, type DiffScorecardOptions, type DirEntry, type DiscoverPersonasOptions, type DiscoveredPersona, DriverResult, DriverState, DualAgentBench, type DualAgentBenchConfig, type DualAgentReport, type DualAgentRound, type DualAgentScenario, type DualAgentScenarioResult, ERROR_COUNT_PATTERNS, type EnsembleAggregate, type EnsembleJudgeOptions, type ErrorCountPattern, type ErrorStreakOptions, type EvalToolDef, EvalTraceStore, type EvolutionRound, type ExecutorConfig, type Expectation, type Experiment, type ExperimentProvenance, type ExperimentRep, type ExperimentStats, type ExperimentStore, ExperimentTracker, type ExperimentTrackerOptions, type ExperimentVerdict, type ExportedRewardModel, type ExtractOptions, type ExtractResult, type FactorContribution, type FactorialCell, FailureClass, FeedbackLabel, FeedbackTrajectory, FeedbackTrajectoryStore, type FieldAgreementSpec, type FieldDestination, type FileChange, type FlowAction, type FlowLayerEnv, type FlowLayerFactoryInput, type FlowRunner, type FlowRunnerStepResult, type FlowSpec, type FlowStep, type GhCliClientOptions, type GoldScenario, type GoldSplit, type GoldenSeverity, type GoldenSpec, HARNESS_NATIVE_MODEL, type HarnessAdapter, HarnessConfig, type HarnessExperimentConfig, type HarnessExperimentResult, type HarnessIntervention, type HarnessRunRequest, type HarnessRunResult, type HarnessScenario, type HarnessSelection, type HarnessVariant, type HarnessVariantReport, type HeadroomClass, type HeadroomInput, type HeldOutPartition, type HiddenCriteriaGrader, type HiddenGradeResult, type HiddenLeak, HoldoutAuditor, type HttpGithubClientOptions, INTENT_MATCH_JUDGE_VERSION, type ImageData, type ImprovementThresholds, type ImprovementVerdictResult, InMemoryWorkspaceInspector, type InferenceScorer, type InspectorContext, type IntentMatchInput, type IntentMatchOptions, type IntentMatchResult, type InteractionContribution, JudgeError, type JudgeFamily, type JudgeFleetOptions, JudgeFn, JudgeParseError, type JudgeReplayResult, type JudgeRetryOutcome, type JudgeRetryPolicy, JudgeRunner, type JudgeScoreInput, type JudgeVerdict, type KeywordConceptSpec, type KeywordCoverageFinding, type KeywordCoverageOptions, type KeywordCoverageResult, type KnowledgeAcquisitionMode, type KnowledgeBundle, type KnowledgeFallbackPolicy, type KnowledgeFreshness, type KnowledgeImportance, type KnowledgeReadinessReport, type KnowledgeRecommendedAction, type KnowledgeRequirement, type KnowledgeRequirementCategory, type KnowledgeResponsibleSurface, type KnowledgeSensitivity, type LangfuseEnvelope, type LangfuseGeneration, type LangfuseScore, Layer, LayerResult, type LeaderboardOptions, type LeaderboardRow, type LiveProofArtifact, type LiveProofConfig, type LiveProofContext, type LiveProofResult, LlmClientOptions, LlmSpan, LockedJsonlAppender, MODEL_PRICING, type MakeEvalToolsConfig, type MatchResult, type MatchedPair, type MatcherResult, McNemarResult, type MeasurementPolicy, type MergeOptions, MetricsCollector, type ModelCostRollup, type ModelPreflight, type ModelSeats, ModelsUnreachableError, type MuffledFinder, type MuffledFinding, type MultiToolchainLayerConfig, type Mutator, Mutex, type NoLeakOptions, type NoProgressOptions, Objective, type Oracle, type OracleObservation, type OracleReport, type OracleResult, type OrthogonalityInput, type OrthogonalityResult, OtelExportConfig, OtelExporter, type OtelPipelineHandle, type OtelPipelineOptions, type PairArmsOptions, type PairArmsResult, type PairedArmRow, type PairedArmsComparison, PairedBootstrapOptions, PairedBootstrapResult, type PairedCorrectness, type PairedMetricDelta, PairwiseSteeringOptimizer, type ParaphraseRobustnessScenarioInput, type ParaphraseRobustnessScenarioResult, ParetoResult, type ParseStudentLabel, type PartitionHeldOutOptions, PersonaConfig, type Playbook, type PlaybookEntry, type PoolSlot, type PrReviewAuditCase, type PrReviewBenchmarkSummary, type PrReviewComment, type PrReviewMatchedFinding, type PrReviewOutcome, type PrReviewReferenceFinding, type PrReviewScore, type PrReviewScoreWeights, type PrReviewSeverity, type PrReviewSource, type PreflightModelsOptions, type PreflightOutcome, type ProductBenchmarkArm, type ProductBenchmarkArtifactPaths, type ProductBenchmarkBudgets, type ProductBenchmarkExportOptions, type ProductBenchmarkExportResult, type ProductBenchmarkManifest, type ProductBenchmarkProfileRef, type ProductBenchmarkRecord, type ProductBenchmarkRepoRef, type ProductBenchmarkRunInput, type ProductBenchmarkScenario, type ProductBenchmarkSingleRunExportOptions, type ProductBenchmarkSplit, type ProductBenchmarkSubstrateVersions, type ProductBenchmarkValidationReport, ProductClient, ProductClientConfig, type ProfileAxisSpec, type PromptHandle, PromptRegistry, type ProposeAutomatedPullRequestInput, type ProposeAutomatedPullRequestResult, type ProvenanceReader, type RecordRunsOptions, type ReferenceMatchResult, type ReferenceReplayAdapter, type ReferenceReplayAdapterFn, type ReferenceReplayAdapterLike, type ReferenceReplayAggregate, type ReferenceReplayCandidate, type ReferenceReplayCase, type ReferenceReplayCaseRun, type ReferenceReplayExecutionScenario, type ReferenceReplayItem, type ReferenceReplayMatch, type ReferenceReplayMatchStrategy, type ReferenceReplayMatcher, type ReferenceReplayPromotionDecision, type ReferenceReplayPromotionPolicy, type ReferenceReplayRun, type ReferenceReplayRunContext, type ReferenceReplayRunOptions, type ReferenceReplayRunStore, type ReferenceReplayScenario, type ReferenceReplayScenarioScore, type ReferenceReplayScore, type ReferenceReplayScoreOptions, type ReferenceReplaySplit, type ReferenceReplaySplitComparison, type ReferenceReplaySteeringRowsOptions, type ReflectionContext, type ReflectionProposal, ReleaseConfidenceScorecard, ReleaseConfidenceThresholds, type RenderStudentPrompt, type RepeatedActionOptions, type RepoRef, type ReviewerMemoryEntry, type ReviewerOutput, type ReviewerPromptInput, type ReviewerSoftFailDefaults, type ReviewerVerificationSummary, RiskDifferenceResult, type RobustnessResult, type RoutedField, Run, type RunCommandInput, type RunCommandResult, type RunDistillationOptions, type RunDistillationResult, RunFilter, RunRecord, type RunRecordBackend, type RunRecordFilter, RunScore, RunScoreWeights, RunSplitTag, RunTrace, type RuntimeResolution, SandboxDriver, SandboxHarnessResult, type SandboxJudgeKind, type SandboxJudgeResult, type SandboxJudgeSpec, type SandboxPool, type ScanOptions, Scenario, type ScenarioCost, ScenarioFile, ScenarioRegistry, ScenarioResult, type ScoreKnowledgeReadinessOptions, type Scorecard, type ScorecardCell, type ScorecardCellDiff, type ScorecardDiff, type ScorecardEntry, type ScorecardLogLine, type ScoredTarget, type SeatName, type SeatPresetName, SeatUnsetError, type SelfPlayOptions, type SelfPlayProposer, type SelfPlayScorer, type SerializedRegex, Severity, type SingleBackendDivergence, SingleBackendError, type SingleBackendField, type SingleBackendReport, type Slo, type SloCheckResult, type SloComparator, type SloReport, type SloSeverity, type SlopCategory, type SlotFactory, Span, type SpanPredicate, type SplitGoldOptions, type SteeringBundle, type SteeringDelta, type SteeringOptimizationResult, type SteeringOptimizationRow, type SteeringOptimizationSelector, type SteeringOptimizerBackend, type SteeringOptimizerConfig, type SteeringRolePrompt, type StepAttribution, type StreamingDetector, type SynthesisReason, type SynthesisTarget, type TaskHeadroom, TestResult, type TextMatcher, type ThresholdContract, TokenCounter, type TokenSpec, type ToolMatcher, ToolSpan, TraceAnalystSpan, type TraceContract, TraceContractBuilder, TraceEmitter, TraceStore, type TracedAnalystOptions, type TracedJudgeOptions, Trajectory, TrajectoryStep, type TreatmentClass, type TreatmentGate, type TreatmentGateInput, type TreatmentGateOptions, type TrialTrace, TurnMetrics, UI_FINDING_SEVERITIES, UI_LENSES, UNIVERSAL_FINDERS, type UiFinding, type UiFindingScreenshot, type UiFindingSeverity, type UiLens, type UserQuestion, type VerdictCacheStats, type VerdictCacheStore, VerifyContext, type VisualDiffOptions, type VisualDiffResult, type ViteDeployRunnerInput, type WorkerDriverContext, type WorkflowTopology, type WorkspaceAssertion, type WorkspaceAssertionResult, type WorkspaceInspector, type WorkspaceSnapshot, type WranglerDeployRunnerInput, acquisitionPlansForKnowledgeGaps, adversarialJudge, agentProfileHash, agentProfileId, agentProfileModelId, agentVisibleFields, aggregateJudgeVerdicts, aggregatePrReviewScore, analyzeAntiSlop, appendScorecard, assertCapabilityHeadroom, assertCrossFamily, assertModelsServed, assertNoHiddenLeak, assertProductBenchmarkRun, assertSingleBackend, assignHeldOutTag, attachCostToReport, attest, attributeCounterfactuals, bisect, blendHeldout, blockingKnowledgeEval, buildAgreementJudge, buildDriverSystemPrompt, buildProductBenchmarkManifest, buildReflectionPrompt, buildReviewerPrompt, buildWorkerDriverSystemPrompt, cachedJudge, canaryLeakView, canonicalJson, capabilityHeadroom, causalAttribution, checkBehavioralCanary, checkCanaries, checkSlos, checkTraceContracts, classifyTreatment, codeExecutionJudge, coherenceJudge, collectionPreserved, commentsForSource, commitBisect, comparePairedArms, compareReferenceReplay, compilerJudge, computeExperimentStats, contentHash, contractJudge, costReport, createAntiSlopJudge, createCustomJudge, createDefaultReviewer, createDomainExpertJudge, createIntentMatchJudge, createSandboxPool, crossTraceDiff, dataDescriptionBits, decideNextUserTurn, decideReferenceReplayPromotion, decideReferenceReplayRunPromotion, defaultBlendWeights, defaultJudges, defaultParseStudentLabel, defaultReferenceReplayMatcher, defaultRenderStudentPrompt, deployGateLayer, diffScorecard, discoverPersonas, distillPlaybook, ensembleJudge, errorStreakDetector, estimateCost, estimateTokens, evaluateContract, evaluateOracles, evaluateTraceContract, executeScenario, expandProfileAxes, expectAgent, exportProductBenchmark, exportProductBenchmarkRuns, exportRewardModel, extractAssetUrls, extractErrorCount, fieldAgreement, fileContains, fileExists, fileExperimentStore, fileVerdictCache, findAutoMatchNoExpectation, findConstructorCwdDropped, findFallbackToPass, findLiteralTruePass, findProductBenchmarkArtifacts, findSkipCountsAsPass, flowLayer, fnv1a32, formatBenchmarkReport, formatDriverReport, formatFindings, formatScorecardDiff, gateTreatmentApplied, gateTreatmentFromMetrics, gateTreatmentFromSpans, gateTreatmentFromToolSpans, ghCliClient, gitProvenanceReader, precision as goldenPrecision, gradeOnHidden, harnessAxisOf, hashContent, hashToUnit, hiddenGrade, htmlContainsElement, httpGithubClient, improvementVerdict, inMemoryExperimentStore, inMemoryReferenceReplayStore, inMemoryRunRecordBackend, inMemoryVerdictCache, isHiddenDestination, isModelPriced, isOtelConfigured, jsonShape, jsonlReferenceReplayStore, jsonlRunRecordBackend, judgeFamily, keyPreserved, knowledgeReadinessTracePayload, leaderboard, linterJudge, loadGoldScenarios, loadScorecard, loadScorerFromGrader, localCommandRunner, lowercaseMutator, makeEvalTools, matchGoldens, matchSpan, mergeLayerResults, mergeSteeringBundle, modelDescriptionBits, multiToolchainLayer, noProgressDetector, notBlocked, observeAll, pairArms, paraphraseRobustness, paraphraseRobustnessScenarios, parseGoldJsonl, parseReflectionResponse, partitionHeldOut, passOrthogonality, pixelDeltaRatio, politenessPrefixMutator, preflightModels, printDriverSummary, productBenchmarkIntegrityFailures, productBenchmarkMutableSurfaces, productBenchmarkRepoIdentity, productBenchmarkSplits, index as profile, promptBisect, proposeSynthesisTargets, readProductBenchmarkManifest, readProductBenchmarkRecords, recordRuns, recordRunsToScorecard, referenceReplayRunsToSteeringRows, referenceReplayScenarioToRunScore, regexMatches, renderMarkdownReport, renderPlaybookMarkdown, renderSteeringText, repeatedActionDetector, replayScorerOverCorpus, replayTraceThroughJudge, resolveModelPricing, resolveSeat, routeFields, rowCount, rowWhere, runAssertions, runBehavioralCanaries, runCanaries, runCounterfactual, runDistillation, runE2EWorkflow, runExpectations, runHarnessExperiment, runIntentMatchJudge, runJudgeFleet, runKeywordCoverageJudge, runKeywordCoverageJudgeUrl, runLiveProof, runRecordToProductBenchmarkRecord, runReferenceReplay, runScore, runSelfPlay, scanForMuffledGates, scoreContinuity, scoreKnowledgeReadiness, scorePrReviewComments, scorePrReviewSource, scoreReferenceReplay, seatPresets, securityJudge, selectHarnessVariant, sentenceReorderMutator, splitGold, statusAdvanced, summarizeHarnessResults, summarizePrReviewBenchmark, testJudge, textInSnapshot, toLangfuseEnvelope, toOpenAiTool, toPrometheusText, traceContract, traceJudge, traceJudgeEnsemble, tracedAnalyzeTraces, typoMutator, urlContains, userQuestionsForKnowledgeGaps, validateProductBenchmarkManifest, validateProductBenchmarkRecord, validateProductBenchmarkRun, verifyAttestation, visualDiff, viteDeployRunner, weightedRecall, whitespaceCollapseMutator, withHeldoutBlend, withJudgeRetry, withOtelPipeline, wranglerDeployRunner };
|