@tangle-network/agent-eval 0.79.0 → 0.81.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +14 -0
- package/README.md +101 -169
- package/dist/adapters/http.d.ts +2 -2
- package/dist/adapters/langchain.d.ts +2 -2
- package/dist/adapters/otel.d.ts +4 -4
- package/dist/{agent-profile-aSEaJ9Pl.d.ts → agent-profile-D0PBIWlV.d.ts} +14 -2
- package/dist/analyst/index.d.ts +10 -10
- package/dist/analyst/index.js +3 -3
- package/dist/{analyst-t7zZS3TV.d.ts → analyst-C8HHvfJp.d.ts} +1 -1
- package/dist/belief-state/index.d.ts +524 -0
- package/dist/belief-state/index.js +1862 -0
- package/dist/belief-state/index.js.map +1 -0
- package/dist/benchmarks/index.d.ts +2 -2
- package/dist/calibration-Cpr3WaX3.d.ts +101 -0
- package/dist/campaign/index.d.ts +40 -120
- package/dist/campaign/index.js +129 -238
- package/dist/campaign/index.js.map +1 -1
- package/dist/chunk-4DIJWVUT.js +131 -0
- package/dist/chunk-4DIJWVUT.js.map +1 -0
- package/dist/{chunk-RPLZ4OIB.js → chunk-BABOZOSN.js} +7 -4
- package/dist/{chunk-RPLZ4OIB.js.map → chunk-BABOZOSN.js.map} +1 -1
- package/dist/{chunk-IHDHUN2X.js → chunk-CVVHBFGN.js} +3 -3
- package/dist/chunk-CVVHBFGN.js.map +1 -0
- package/dist/{chunk-B26KI423.js → chunk-FZWAFVAA.js} +2 -2
- package/dist/{chunk-ITBRCT73.js → chunk-IDVBLYCY.js} +2 -10
- package/dist/chunk-IDVBLYCY.js.map +1 -0
- package/dist/{chunk-5LVWPNS5.js → chunk-L5G7OUKD.js} +4 -4
- package/dist/{chunk-5LVWPNS5.js.map → chunk-L5G7OUKD.js.map} +1 -1
- package/dist/chunk-NPCTHQIO.js +91 -0
- package/dist/chunk-NPCTHQIO.js.map +1 -0
- package/dist/{chunk-GXHLRXDI.js → chunk-OTYQPHPL.js} +4 -4
- package/dist/{chunk-6REHLN5J.js → chunk-QS3RBQPI.js} +2 -2
- package/dist/{chunk-GWGO2K6Y.js → chunk-RBNA5AZT.js} +2 -2
- package/dist/chunk-S42AWHMP.js +697 -0
- package/dist/chunk-S42AWHMP.js.map +1 -0
- package/dist/chunk-VI2UW6B6.js +162 -0
- package/dist/chunk-VI2UW6B6.js.map +1 -0
- package/dist/{chunk-CF67I6QY.js → chunk-VIDQF3F5.js} +2 -2
- package/dist/{chunk-XXNIODOM.js → chunk-WJL2NJXN.js} +3 -3
- package/dist/{chunk-LB2UOI5F.js → chunk-YGYXHNAQ.js} +47 -160
- package/dist/chunk-YGYXHNAQ.js.map +1 -0
- package/dist/{chunk-KX6F6NCG.js → chunk-Z7VFTS2J.js} +2 -2
- package/dist/{chunk-ZPSKPT3V.js → chunk-ZZ2HOPME.js} +5 -2
- package/dist/chunk-ZZ2HOPME.js.map +1 -0
- package/dist/cli.js +2 -2
- package/dist/code-agent-session-BRXmavYv.d.ts +80 -0
- package/dist/contract/index.d.ts +132 -18
- package/dist/contract/index.js +139 -6
- package/dist/contract/index.js.map +1 -1
- package/dist/{control-CehLtoET.d.ts → control-GeE8OhpN.d.ts} +1 -1
- package/dist/control.d.ts +2 -2
- package/dist/governance/index.d.ts +1 -1
- package/dist/hosted/index.d.ts +4 -4
- package/dist/{index-B1RKber3.d.ts → index-DE3RXAXD.d.ts} +1 -1
- package/dist/index.d.ts +79 -288
- package/dist/index.js +87 -410
- package/dist/index.js.map +1 -1
- package/dist/{insight-report-dlpEzQDi.d.ts → insight-report-3ADTfClO.d.ts} +1 -1
- package/dist/{kind-factory-DqV2t1Xk.d.ts → kind-factory-CVecZZG_.d.ts} +2 -2
- package/dist/{llm-client-DbjLfz-K.d.ts → llm-client-CuUg2Mn3.d.ts} +1 -1
- package/dist/meta-eval/index.d.ts +6 -99
- package/dist/meta-eval/index.js +7 -76
- package/dist/meta-eval/index.js.map +1 -1
- package/dist/off-policy-DiwuKKg7.d.ts +132 -0
- package/dist/openapi.json +1 -1
- package/dist/{outcome-store-D6KWmYvj.d.ts → outcome-store-rnXLEqSn.d.ts} +1 -1
- package/dist/pipelines/index.js +2 -2
- package/dist/{provenance-CEAJI9rm.d.ts → provenance-B9Q4886D.d.ts} +4 -4
- package/dist/{registry-BmEuU94S.d.ts → registry-DrEQ3Luj.d.ts} +2 -2
- package/dist/{release-report-CXXZlR8g.d.ts → release-report-hlNtD12q.d.ts} +2 -2
- package/dist/reporting.d.ts +6 -6
- package/dist/reporting.js +3 -3
- package/dist/{researcher-rInLj9De.d.ts → researcher-BLPHBbNV.d.ts} +3 -3
- package/dist/rl.d.ts +11 -141
- package/dist/rl.js +10 -124
- package/dist/rl.js.map +1 -1
- package/dist/{rubric-predictive-validity-CWyWWLBg.d.ts → rubric-predictive-validity-CnEl9Jc8.d.ts} +2 -2
- package/dist/{run-campaign-OVEZF24D.js → run-campaign-4Y5V5CN3.js} +3 -3
- package/dist/{run-improvement-loop-Bgu4C59E.d.ts → run-improvement-loop-D6PZOoQL.d.ts} +2 -2
- package/dist/{run-record-sItO5ftF.d.ts → run-record-De9VarXR.d.ts} +1 -1
- package/dist/{semantic-concept-judge-Du4ZVyef.d.ts → semantic-concept-judge-DIEgr_6v.d.ts} +6 -6
- package/dist/{statistics-B7yCbi9i.d.ts → statistics-CnC1FMbx.d.ts} +3 -7
- package/dist/{store-GmBE2pZZ.d.ts → store-C1YxJDEK.d.ts} +1 -1
- package/dist/{summary-report-BTaXq1TS.d.ts → summary-report-Db0dDSWP.d.ts} +1 -1
- package/dist/traces.d.ts +5 -5
- package/dist/{types-DRvV0zRo.d.ts → types-Cu3u_x59.d.ts} +3 -3
- package/dist/{types-QHG0KnkF.d.ts → types-D7lLRYe9.d.ts} +2 -2
- package/dist/wire/index.js +2 -2
- package/dist/workflow/index.d.ts +7 -7
- package/dist/workflow/index.js +1 -1
- package/docs/concepts.md +1 -0
- package/docs/research/belief-state-agent-eval-roadmap.md +590 -0
- package/docs/research/research-roadmap.md +1 -0
- package/docs/self-improvement-map.md +111 -0
- package/package.json +7 -2
- package/dist/chunk-IHDHUN2X.js.map +0 -1
- package/dist/chunk-ITBRCT73.js.map +0 -1
- package/dist/chunk-LB2UOI5F.js.map +0 -1
- package/dist/chunk-ZPSKPT3V.js.map +0 -1
- /package/dist/{chunk-B26KI423.js.map → chunk-FZWAFVAA.js.map} +0 -0
- /package/dist/{chunk-GXHLRXDI.js.map → chunk-OTYQPHPL.js.map} +0 -0
- /package/dist/{chunk-6REHLN5J.js.map → chunk-QS3RBQPI.js.map} +0 -0
- /package/dist/{chunk-GWGO2K6Y.js.map → chunk-RBNA5AZT.js.map} +0 -0
- /package/dist/{chunk-CF67I6QY.js.map → chunk-VIDQF3F5.js.map} +0 -0
- /package/dist/{chunk-XXNIODOM.js.map → chunk-WJL2NJXN.js.map} +0 -0
- /package/dist/{chunk-KX6F6NCG.js.map → chunk-Z7VFTS2J.js.map} +0 -0
- /package/dist/{run-campaign-OVEZF24D.js.map → run-campaign-4Y5V5CN3.js.map} +0 -0
|
@@ -3,7 +3,7 @@ import { C as ControlEvalResult, a as ControlRunResult, h as ControlRuntimeConfi
|
|
|
3
3
|
import { T as TraceEmitter } from './emitter-DEZwY14K.js';
|
|
4
4
|
import { F as FailureClass } from './schema-m0gsnbt3.js';
|
|
5
5
|
import { T as TraceStore } from './store-CKUAgsJz.js';
|
|
6
|
-
import {
|
|
6
|
+
import { a as RunSplitTag, b as RunTokenUsage, R as RunRecord } from './run-record-De9VarXR.js';
|
|
7
7
|
|
|
8
8
|
interface ActionExecutionPolicy {
|
|
9
9
|
allowedTypes?: string[];
|
package/dist/control.d.ts
CHANGED
|
@@ -1,4 +1,4 @@
|
|
|
1
|
-
export { A as ActionExecutionPolicy, a as ActionPolicyDecision, C as ControlRunToRunRecordOptions, e as ProposeReviewConfig, f as ProposeReviewControlAction, g as ProposeReviewControlConfig, h as ProposeReviewControlResult, i as ProposeReviewControlState, j as ProposeReviewReport, p as RunEvidenceMetadata, s as controlRunToRunRecord, u as evaluateActionPolicy, x as runProposeReview, y as runProposeReviewAsControlLoop, z as scoreFromEvals } from './control-
|
|
1
|
+
export { A as ActionExecutionPolicy, a as ActionPolicyDecision, C as ControlRunToRunRecordOptions, e as ProposeReviewConfig, f as ProposeReviewControlAction, g as ProposeReviewControlConfig, h as ProposeReviewControlResult, i as ProposeReviewControlState, j as ProposeReviewReport, p as RunEvidenceMetadata, s as controlRunToRunRecord, u as evaluateActionPolicy, x as runProposeReview, y as runProposeReviewAsControlLoop, z as scoreFromEvals } from './control-GeE8OhpN.js';
|
|
2
2
|
export { c as ControlActionFailureMode, d as ControlActionOutcome, e as ControlBudget, f as ControlContext, g as ControlDecision, C as ControlEvalResult, a as ControlRunResult, h as ControlRuntimeConfig, i as ControlRuntimeError, j as ControlSeverity, b as ControlStep, k as ControlStopPolicies, S as StopDecision, l as allCriticalPassed, o as objectiveEval, r as runAgentControlLoop, s as stopOnNoProgress, m as stopOnRepeatedAction, n as subjectiveEval } from './control-runtime-DuFBYg7A.js';
|
|
3
3
|
import './feedback-trajectory-B3rErRsh.js';
|
|
4
4
|
import './dataset-B2kL-fSM.js';
|
|
@@ -6,4 +6,4 @@ import './errors-Dwqw-T_m.js';
|
|
|
6
6
|
import './emitter-DEZwY14K.js';
|
|
7
7
|
import './schema-m0gsnbt3.js';
|
|
8
8
|
import './store-CKUAgsJz.js';
|
|
9
|
-
import './run-record-
|
|
9
|
+
import './run-record-De9VarXR.js';
|
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
import { c as DatasetManifest } from '../dataset-B2kL-fSM.js';
|
|
2
2
|
import { b as CalibrationResult } from '../judge-calibration-DilmB3Ml.js';
|
|
3
|
-
import {
|
|
3
|
+
import { b as OutcomeStore } from '../outcome-store-rnXLEqSn.js';
|
|
4
4
|
import { d as RedTeamReport } from '../red-team-DW9Ca_tj.js';
|
|
5
5
|
import { T as TraceStore } from '../store-CKUAgsJz.js';
|
|
6
6
|
import '../errors-Dwqw-T_m.js';
|
package/dist/hosted/index.d.ts
CHANGED
|
@@ -1,9 +1,9 @@
|
|
|
1
|
-
import { M as MutableSurface,
|
|
2
|
-
import { I as InsightReport } from '../insight-report-
|
|
3
|
-
import '../run-record-
|
|
1
|
+
import { M as MutableSurface, c as GateDecision } from '../types-D7lLRYe9.js';
|
|
2
|
+
import { I as InsightReport } from '../insight-report-3ADTfClO.js';
|
|
3
|
+
import '../run-record-De9VarXR.js';
|
|
4
4
|
import '../errors-Dwqw-T_m.js';
|
|
5
5
|
import '../schema-m0gsnbt3.js';
|
|
6
|
-
import '../summary-report-
|
|
6
|
+
import '../summary-report-Db0dDSWP.js';
|
|
7
7
|
import '../failure-cluster-CL7IVgkJ.js';
|
|
8
8
|
import '../store-CKUAgsJz.js';
|
|
9
9
|
import '../judge-calibration-DilmB3Ml.js';
|
package/dist/index.d.ts
CHANGED
|
@@ -1,11 +1,11 @@
|
|
|
1
|
-
export { A as ActionExecutionPolicy, a as ActionPolicyDecision, C as ControlRunToRunRecordOptions, L as LlmJsonCall, b as LlmReviewerConfig, P as ProposeFn, c as ProposeInput, d as ProposeOutput, e as ProposeReviewConfig, f as ProposeReviewControlAction, g as ProposeReviewControlConfig, h as ProposeReviewControlResult, i as ProposeReviewControlState, j as ProposeReviewReport, k as ProposeReviewShot, R as Review, l as ReviewFn, m as ReviewInput, n as ReviewMemoryEntry, o as ReviewMemoryStore, p as RunEvidenceMetadata, V as Verification, q as VerifyFn, r as controlFailureClassFromVerification, s as controlRunToRunRecord, t as createLlmReviewer, u as evaluateActionPolicy, v as inMemoryReviewStore, w as jsonlReviewStore, x as runProposeReview, y as runProposeReviewAsControlLoop, z as scoreFromEvals } from './control-
|
|
2
|
-
import { R as RunRecord } from './run-record-
|
|
3
|
-
export { e as AGENT_PROFILE_KINDS, A as AgentProfileCell, d as AgentProfileCellInput, f as AgentProfileCellSchemaVersion, g as AgentProfileCellValidationError, h as AgentProfileDimensionValue, i as AgentProfileHarness, j as AgentProfileJson, k as AgentProfileKind, l as AgentProfileSource, m as AgentProfileSourceInput, J as JudgeScoresRecord, c as RunJudgeMetadata, n as RunOutcome, o as RunRecordValidationError,
|
|
4
|
-
export { B as BehavioralMetrics, z as ConceptComplexity, A as ConceptFinding, E as ConceptSpec, G as ConceptWeightStrategy, C as CreateAnalystAiConfig, H as DEFAULT_COMPLEXITY_WEIGHTS, D as DEFAULT_TRACE_ANALYST_KINDS, b as DefaultAnalystRegistryOptions, c as DiffPolicy, F as FAILURE_MODE_KIND_SPEC, f as FindingSubject, g as FindingSubjectKind, i as FindingsDiff, j as FindingsStore, I as IMPROVEMENT_KIND_SPEC, k as KNOWLEDGE_GAP_KIND_SPEC, l as KNOWLEDGE_POISONING_KIND_SPEC, P as PersistedFinding, J as SEMANTIC_CONCEPT_JUDGE_VERSION, m as SKILL_USAGE_ANALYST, a as SemanticConceptJudgeInput, S as SemanticConceptJudgeOptions, L as SemanticConceptJudgeResult, n as SkillUsageAnalyst, M as SuboptimalCode, N as SuboptimalSignal, r as buildDefaultAnalystRegistry, O as computeTraceMetrics, t as createAnalystAi, Q as createSemanticConceptJudge, u as defaultIsMaterial, v as diffFindings, R as runSemanticConceptJudge } from './semantic-concept-judge-
|
|
5
|
-
import { l as ChatRequest, p as CreateChatClientOpts } from './types-
|
|
6
|
-
export { A as Analyst, a as AnalystContext, g as AnalystCost, c as AnalystFinding, i as AnalystInputKind, j as AnalystRequirements, f as AnalystRunEvent, e as AnalystRunInputs, d as AnalystRunResult, b as AnalystRunSummary, h as AnalystSeverity, k as ChatCallOpts, C as ChatClient, m as ChatResponse, n as ChatTransport, o as CliBridgeTransportOpts, D as DirectProviderTransportOpts, E as EvidenceRef, M as MockTransportOpts, R as RouterTransportOpts, S as SandboxSdkTransportOpts, q as computeFindingId, r as createChatClient, s as makeFinding } from './types-
|
|
7
|
-
export { C as CreateTraceAnalystKindOpts, a as RawAnalystFinding, T as TraceAnalystGolden, c as TraceAnalystKindSpec, d as createTraceAnalystKind, r as renderPriorFindings } from './kind-factory-
|
|
8
|
-
export {
|
|
1
|
+
export { A as ActionExecutionPolicy, a as ActionPolicyDecision, C as ControlRunToRunRecordOptions, L as LlmJsonCall, b as LlmReviewerConfig, P as ProposeFn, c as ProposeInput, d as ProposeOutput, e as ProposeReviewConfig, f as ProposeReviewControlAction, g as ProposeReviewControlConfig, h as ProposeReviewControlResult, i as ProposeReviewControlState, j as ProposeReviewReport, k as ProposeReviewShot, R as Review, l as ReviewFn, m as ReviewInput, n as ReviewMemoryEntry, o as ReviewMemoryStore, p as RunEvidenceMetadata, V as Verification, q as VerifyFn, r as controlFailureClassFromVerification, s as controlRunToRunRecord, t as createLlmReviewer, u as evaluateActionPolicy, v as inMemoryReviewStore, w as jsonlReviewStore, x as runProposeReview, y as runProposeReviewAsControlLoop, z as scoreFromEvals } from './control-GeE8OhpN.js';
|
|
2
|
+
import { R as RunRecord } from './run-record-De9VarXR.js';
|
|
3
|
+
export { e as AGENT_PROFILE_KINDS, A as AgentProfileCell, d as AgentProfileCellInput, f as AgentProfileCellSchemaVersion, g as AgentProfileCellValidationError, h as AgentProfileDimensionValue, i as AgentProfileHarness, j as AgentProfileJson, k as AgentProfileKind, l as AgentProfileSource, m as AgentProfileSourceInput, J as JudgeScoresRecord, c as RunJudgeMetadata, n as RunOutcome, o as RunRecordValidationError, a as RunSplitTag, b as RunTokenUsage, S as SandboxAgentProfileLike, p as agentProfileCellHashMaterial, q as agentProfileCellKey, r as assertRunAgentProfileCell, s as buildAgentProfileCell, t as buildSandboxAgentProfileCell, u as groupRunsByAgentProfileCell, v as isRunRecord, w as parseRunRecordSafe, x as requireAgentProfileCell, y as roundTripRunRecord, z as toAgentProfileJson, B as validateAgentProfileCell, C as validateRunRecord, D as verifyAgentProfileCell } from './run-record-De9VarXR.js';
|
|
4
|
+
export { B as BehavioralMetrics, z as ConceptComplexity, A as ConceptFinding, E as ConceptSpec, G as ConceptWeightStrategy, C as CreateAnalystAiConfig, H as DEFAULT_COMPLEXITY_WEIGHTS, D as DEFAULT_TRACE_ANALYST_KINDS, b as DefaultAnalystRegistryOptions, c as DiffPolicy, F as FAILURE_MODE_KIND_SPEC, f as FindingSubject, g as FindingSubjectKind, i as FindingsDiff, j as FindingsStore, I as IMPROVEMENT_KIND_SPEC, k as KNOWLEDGE_GAP_KIND_SPEC, l as KNOWLEDGE_POISONING_KIND_SPEC, P as PersistedFinding, J as SEMANTIC_CONCEPT_JUDGE_VERSION, m as SKILL_USAGE_ANALYST, a as SemanticConceptJudgeInput, S as SemanticConceptJudgeOptions, L as SemanticConceptJudgeResult, n as SkillUsageAnalyst, M as SuboptimalCode, N as SuboptimalSignal, r as buildDefaultAnalystRegistry, O as computeTraceMetrics, t as createAnalystAi, Q as createSemanticConceptJudge, u as defaultIsMaterial, v as diffFindings, R as runSemanticConceptJudge } from './semantic-concept-judge-DIEgr_6v.js';
|
|
5
|
+
import { l as ChatRequest, p as CreateChatClientOpts } from './types-Cu3u_x59.js';
|
|
6
|
+
export { A as Analyst, a as AnalystContext, g as AnalystCost, c as AnalystFinding, i as AnalystInputKind, j as AnalystRequirements, f as AnalystRunEvent, e as AnalystRunInputs, d as AnalystRunResult, b as AnalystRunSummary, h as AnalystSeverity, k as ChatCallOpts, C as ChatClient, m as ChatResponse, n as ChatTransport, o as CliBridgeTransportOpts, D as DirectProviderTransportOpts, E as EvidenceRef, M as MockTransportOpts, R as RouterTransportOpts, S as SandboxSdkTransportOpts, q as computeFindingId, r as createChatClient, s as makeFinding } from './types-Cu3u_x59.js';
|
|
7
|
+
export { C as CreateTraceAnalystKindOpts, a as RawAnalystFinding, T as TraceAnalystGolden, c as TraceAnalystKindSpec, d as createTraceAnalystKind, r as renderPriorFindings } from './kind-factory-CVecZZG_.js';
|
|
8
|
+
export { A as AnalystHooks, a as AnalystRegistry, b as AnalystRegistryOptions, B as BudgetPolicy, R as RegistryRunOpts } from './registry-DrEQ3Luj.js';
|
|
9
9
|
import { TCloud } from '@tangle-network/tcloud';
|
|
10
10
|
import { B as BenchmarkRunnerConfig, S as Scenario, c as BenchmarkReport, P as ProductClientConfig, C as CheckResult, T as TestResult, d as PersonaConfig, D as DriverResult, e as DriverState, b as JudgeFn, f as CollectedArtifacts, g as ScenarioResult, h as TurnMetrics, i as ScenarioFile, j as CompletionCriterion } from './types-Croy5h7V.js';
|
|
11
11
|
export { A as ArtifactCheck, k as ArtifactResult, E as EvalResult, F as FeedbackPattern, l as JudgeConfig, a as JudgeInput, m as JudgeRubric, J as JudgeScore, n as PersonaRigor, R as RouteMap, o as RubricDimension, p as Turn, q as TurnResult } from './types-Croy5h7V.js';
|
|
@@ -14,30 +14,30 @@ import { A as AgentEvalError } from './errors-Dwqw-T_m.js';
|
|
|
14
14
|
export { a as AgentEvalErrorCode, C as CaptureIntegrityError, b as ConfigError, J as JudgeError, N as NotFoundError, R as ReplayError, V as ValidationError, c as VerificationError } from './errors-Dwqw-T_m.js';
|
|
15
15
|
import { b as FeedbackLabel, F as FeedbackTrajectoryStore, a as FeedbackTrajectory } from './feedback-trajectory-B3rErRsh.js';
|
|
16
16
|
export { c as FeedbackArtifactType, d as FeedbackAttempt, e as FeedbackLabelKind, f as FeedbackLabelSource, g as FeedbackOptimizerRow, h as FeedbackOutcome, i as FeedbackReplayAdapter, j as FeedbackReplayResult, k as FeedbackSeverity, l as FeedbackSplitPolicy, m as FeedbackTask, n as FeedbackTrajectoryFilter, o as FileSystemFeedbackTrajectoryStore, I as InMemoryFeedbackTrajectoryStore, P as PreferenceMemoryEntry, p as ProposedSideEffect, q as assignFeedbackSplit, r as controlRunToFeedbackTrajectory, s as createFeedbackTrajectory, t as feedbackTrajectoriesToDatasetScenarios, u as feedbackTrajectoriesToOptimizerRows, v as feedbackTrajectoryToDatasetScenario, w as feedbackTrajectoryToOptimizerRow, x as parseFeedbackTrajectoriesJsonl, y as renderPreferenceMemoryMarkdown, z as replayFeedbackTrajectories, A as replayFeedbackTrajectory, B as serializeFeedbackTrajectoriesJsonl, C as summarizePreferenceMemory, D as withAssignedFeedbackSplit } from './feedback-trajectory-B3rErRsh.js';
|
|
17
|
-
import { A as AgentProfile$1 } from './agent-profile-
|
|
18
|
-
export { c as ArtifactCheckArtifact, d as ArtifactEventLike, e as ArtifactValidator, f as BackendIntegrityError, B as BackendIntegrityReport, C as CompletionRequirement, a as CompletionVerdict, b as CorrectnessChecker, L as LlmCorrectnessCheckerOpts, g as ProducedProposal, P as ProducedState, h as ProposalEventLike, i as RequirementCheck, R as RuntimeEventLike, S as SatisfiedBy, T as TaskGold, j as ToolCallEventLike, V as ValidationContext, k as ValidationIssue, l as ValidationResult, m as agentProfileHash, n as assertRealBackend, o as byteLengthRange, p as composeValidators, q as containsAll, r as createLlmCorrectnessChecker, s as
|
|
17
|
+
import { A as AgentProfile$1 } from './agent-profile-D0PBIWlV.js';
|
|
18
|
+
export { c as ArtifactCheckArtifact, d as ArtifactEventLike, e as ArtifactValidator, f as BackendIntegrityError, B as BackendIntegrityReport, C as CompletionRequirement, a as CompletionVerdict, b as CorrectnessChecker, L as LlmCorrectnessCheckerOpts, g as ProducedProposal, P as ProducedState, h as ProposalEventLike, i as RequirementCheck, R as RuntimeEventLike, S as SatisfiedBy, T as TaskGold, j as ToolCallEventLike, V as ValidationContext, k as ValidationIssue, l as ValidationResult, m as agentProfileHash, n as assertRealBackend, o as byteLengthRange, p as composeValidators, q as containsAll, r as createLlmCorrectnessChecker, s as createTokenRecallChecker, t as extractProducedState, u as jsonHasKeys, v as parseCorrectnessResponse, w as regexMatch, x as summarizeBackendIntegrity, y as verifyCompletion } from './agent-profile-D0PBIWlV.js';
|
|
19
19
|
export { DataAcquisitionPlan, KnowledgeAcquisitionMode, KnowledgeBundle, KnowledgeFallbackPolicy, KnowledgeFreshness, KnowledgeImportance, KnowledgeReadinessReport, KnowledgeRecommendedAction, KnowledgeRequirement, KnowledgeRequirementCategory, KnowledgeResponsibleSurface, KnowledgeSensitivity, ScoreKnowledgeReadinessOptions, UserQuestion, acquisitionPlansForKnowledgeGaps, blockingKnowledgeEval, knowledgeReadinessTracePayload, scoreKnowledgeReadiness, userQuestionsForKnowledgeGaps } from './knowledge/index.js';
|
|
20
|
-
import { h as ReleaseConfidenceThresholds, f as ReleaseConfidenceScorecard } from './release-report-
|
|
21
|
-
export { A as ActionableSideInfo, o as AsiSeverity, B as BootstrapOptions, a as BootstrapResult, J as JudgeReplayGateArgs, R as ReleaseConfidenceAxis, b as ReleaseConfidenceAxisName, c as ReleaseConfidenceInput, d as ReleaseConfidenceIssue, e as ReleaseConfidenceMetrics, g as ReleaseConfidenceStatus, i as ReleaseTraceEvidence, j as RenderReleaseReportOptions, V as Verdict, k as assertReleaseConfidence, l as bootstrapCi, m as evaluateReleaseConfidence, n as judgeReplayGate, r as renderReleaseReport } from './release-report-
|
|
22
|
-
export { C as CliffsMagnitude, c as CorpusAgreementOptions, d as CorpusAgreementPerDimension, e as CorpusAgreementReport, f as CorpusScoreRecord, P as PairedBootstrapOptions, a as PairedBootstrapResult, W as WeightedCompositeInput, g as WeightedCompositeResult, b as benjaminiHochberg, h as bonferroni, i as cliffsDelta, j as cohensD, k as confidenceInterval, l as corpusInterRaterAgreement, m as corpusInterRaterAgreementFromJudgeScores, n as interRaterReliability, o as interpretCliffs, q as mannWhitneyU, r as normalizeScores, p as pairedBootstrap, s as pairedMde, t as pairedTTest, u as partialCredit, v as requiredSampleSize, x as weightedComposite, y as weightedMean, w as wilcoxonSignedRank } from './statistics-
|
|
23
|
-
import { a as AnalyzeTracesInput, A as AnalyzeTracesOptions, b as AnalyzeTracesResult } from './analyst-
|
|
24
|
-
export { c as AnalyzeTracesTurnSnapshot, d as analyzeTraces } from './analyst-
|
|
20
|
+
import { h as ReleaseConfidenceThresholds, f as ReleaseConfidenceScorecard } from './release-report-hlNtD12q.js';
|
|
21
|
+
export { A as ActionableSideInfo, o as AsiSeverity, B as BootstrapOptions, a as BootstrapResult, J as JudgeReplayGateArgs, R as ReleaseConfidenceAxis, b as ReleaseConfidenceAxisName, c as ReleaseConfidenceInput, d as ReleaseConfidenceIssue, e as ReleaseConfidenceMetrics, g as ReleaseConfidenceStatus, i as ReleaseTraceEvidence, j as RenderReleaseReportOptions, V as Verdict, k as assertReleaseConfidence, l as bootstrapCi, m as evaluateReleaseConfidence, n as judgeReplayGate, r as renderReleaseReport } from './release-report-hlNtD12q.js';
|
|
22
|
+
export { C as CliffsMagnitude, c as CorpusAgreementOptions, d as CorpusAgreementPerDimension, e as CorpusAgreementReport, f as CorpusScoreRecord, P as PairedBootstrapOptions, a as PairedBootstrapResult, W as WeightedCompositeInput, g as WeightedCompositeResult, b as benjaminiHochberg, h as bonferroni, i as cliffsDelta, j as cohensD, k as confidenceInterval, l as corpusInterRaterAgreement, m as corpusInterRaterAgreementFromJudgeScores, n as interRaterReliability, o as interpretCliffs, q as mannWhitneyU, r as normalizeScores, p as pairedBootstrap, s as pairedMde, t as pairedTTest, u as partialCredit, v as requiredSampleSize, x as weightedComposite, y as weightedMean, w as wilcoxonSignedRank } from './statistics-CnC1FMbx.js';
|
|
23
|
+
import { a as AnalyzeTracesInput, A as AnalyzeTracesOptions, b as AnalyzeTracesResult } from './analyst-C8HHvfJp.js';
|
|
24
|
+
export { c as AnalyzeTracesTurnSnapshot, d as analyzeTraces } from './analyst-C8HHvfJp.js';
|
|
25
25
|
import { OtelExporter, OtelExportConfig } from './traces.js';
|
|
26
26
|
export { CaptureFetchContext, CaptureFetchOptions, ExportableSpan, FlattenOtlpOptions, OTEL_AGENT_EVAL_SCOPE, OtlpExport, OtlpFileTraceStore, OtlpFileTraceStoreOptions, OtlpFlatLine, OtlpResourceSpans, OtlpSpan, OtlpToRunRecordsOptions, OtlpTraceRunRecord, ProjectedOtlpSpan, ReplayCache, ReplayCacheEntry, ReplayCacheMissError, ReplayCacheStats, ReplayFetchOptions, SpanNotFoundError, TRACE_ANALYST_ACTOR_DESCRIPTION, TRACE_ANALYST_ACTOR_DESCRIPTION_VERSION, TRACE_ANALYST_SUBAGENT_DESCRIPTION, TraceAggregate, TraceAnalystHookOptions, TraceFileMissingError, TraceInsightContext, TraceInsightFinding, TraceInsightPanelRole, TraceInsightPromptInput, TraceInsightQualityGate, TraceInsightQuestion, TraceInsightReadiness, TraceInsightSuite, TraceInsightTask, TraceNotFoundError, TraceStoreSource, TraceStoreToOtlpOptions, TracesToOtlpResult, asNumber, asString, buildTraceAnalystTools, buildTraceInsightContext, buildTraceInsightPrompt, captureFetchToRawSink, convertTraceStoresToOtlp, createOtelExporter, createOtelTracingStore, createReplayFetch, defaultTraceInsightPanel, describeTraceInsightScope, domainEvidencePattern, exportRunAsOtlp, extractOtlpAttributes, firstNumberAttr, firstStringAttr, flattenOtlpExportToNdjson, inferDomainKeywords, inferOtlpKind, iterateRawCalls, otelRunCompleteHook, otlpToRunRecords, otlpToTraceRunRecords, planTraceInsightQuestions, projectOtlpFlatLine, readOtlpStatus, scoreTraceInsightReadiness, stringField, tokenizeDomainWords, traceAnalystFunctionGroup, traceAnalystOnRunComplete } from './traces.js';
|
|
27
|
-
export { D as DEFAULT_TRACE_ANALYST_BUDGETS, b as DatasetOverview, Q as QueryTracesPage, S as SearchSpanResult, c as SearchTraceResult, d as SpanMatchRecord, e as TRACE_ANALYST_TRUNCATION_MARKER_PREFIX, T as TraceAnalysisStore, f as TraceAnalystByteBudgets, g as TraceAnalystFilters, a as TraceAnalystSpan, h as TraceAnalystSpanKind, i as TraceAnalystSpanStatus, j as TraceAnalystTraceSummary, V as ViewSpansResult, k as ViewTraceOversized, l as ViewTraceResult } from './store-
|
|
28
|
-
import { d as RunCriticOptions, a as RunTrace, b as RunScoreWeights, R as RunScore } from './run-critic-BAIjX99r.js';
|
|
29
|
-
export { D as DEFAULT_RUN_SCORE_WEIGHTS, c as RunCritic, e as aggregateRunScore, f as clamp01 } from './run-critic-BAIjX99r.js';
|
|
27
|
+
export { D as DEFAULT_TRACE_ANALYST_BUDGETS, b as DatasetOverview, E as ErrorCluster, Q as QueryTracesPage, S as SearchSpanResult, c as SearchTraceResult, d as SpanMatchRecord, e as TRACE_ANALYST_TRUNCATION_MARKER_PREFIX, T as TraceAnalysisStore, f as TraceAnalystByteBudgets, g as TraceAnalystFilters, a as TraceAnalystSpan, h as TraceAnalystSpanKind, i as TraceAnalystSpanStatus, j as TraceAnalystTraceSummary, V as ViewSpansResult, k as ViewTraceOversized, l as ViewTraceResult } from './store-C1YxJDEK.js';
|
|
30
28
|
import { S as SteeringBundle } from './harness-optimizer-EnEnQPsr.js';
|
|
31
29
|
export { D as DEFAULT_HARNESS_OBJECTIVES, H as HarnessAdapter, a as HarnessExperimentConfig, b as HarnessExperimentResult, c as HarnessIntervention, d as HarnessRunRequest, e as HarnessRunResult, f as HarnessScenario, g as HarnessSelection, h as HarnessVariant, i as HarnessVariantReport, M as MeasurementPolicy, j as SteeringDelta, k as SteeringRolePrompt, W as WorkflowTopology, m as mergeSteeringBundle, r as renderSteeringText, l as runHarnessExperiment, s as selectHarnessVariant, n as summarizeHarnessResults } from './harness-optimizer-EnEnQPsr.js';
|
|
32
30
|
import { S as SandboxDriver, H as HarnessConfig, a as SandboxHarnessResult } from './test-graded-scenario-BdVaPyHT.js';
|
|
33
31
|
export { D as DockerSandboxDriver, c as SandboxHarness, d as SandboxResult, e as SubprocessSandboxDriver, f as SubprocessSandboxDriverOptions, g as TestGradedRunOptions, b as TestGradedRunResult, T as TestGradedScenario, h as TestOutputParser, i as composeParsers, j as jestTestParser, p as pytestTestParser, r as runTestGradedScenario, v as vitestTestParser } from './test-graded-scenario-BdVaPyHT.js';
|
|
32
|
+
import { b as RunScoreWeights, R as RunScore } from './run-critic-BAIjX99r.js';
|
|
33
|
+
export { D as DEFAULT_RUN_SCORE_WEIGHTS, c as RunCritic, d as RunCriticOptions, a as RunTrace, e as aggregateRunScore, f as clamp01 } from './run-critic-BAIjX99r.js';
|
|
34
34
|
import { T as TraceEmitter } from './emitter-DEZwY14K.js';
|
|
35
35
|
export { R as RunCompleteHook, a as RunCompleteHookContext, S as SpanHandle, b as TraceEmitterOptions, l as llmSpanFromProvider } from './emitter-DEZwY14K.js';
|
|
36
36
|
export { b as RunIntegrityError, R as RunIntegrityExpectations, c as RunIntegrityIssue, d as RunIntegrityIssueCode, a as RunIntegrityReport, e as assertRunCaptured, t as throwIfRunIncomplete } from './integrity-CJzrpUua.js';
|
|
37
37
|
export { a as aggregateLlm, b as argHash, g as groupBy, j as judgeSpans, l as llmSpans, r as runFailureClass, c as runsForScenario, t as toolSpans } from './query-CqTxMwDw.js';
|
|
38
38
|
export { F as FileSystemRawProviderSink, a as FileSystemRawProviderSinkOptions, I as InMemoryRawProviderSink, b as InMemoryRawProviderSinkOptions, N as NoopRawProviderSink, P as ProviderRedactor, c as RawProviderDirection, d as RawProviderEvent, R as RawProviderSink, e as RawProviderSinkFilter, f as defaultProviderRedactor, p as providerFromBaseUrl } from './raw-provider-sink-C46HDghv.js';
|
|
39
39
|
export { D as DEFAULT_REDACTION_RULES, b as REDACTION_VERSION, a as RedactionReport, R as RedactionRule, r as redactString, c as redactValue } from './redact-B40YG2M_.js';
|
|
40
|
-
import { h as BudgetSpec, B as BudgetLedgerEntry, R as Run
|
|
40
|
+
import { h as BudgetSpec, B as BudgetLedgerEntry, R as Run, L as LlmSpan } from './schema-m0gsnbt3.js';
|
|
41
41
|
export { A as Artifact, E as EventKind, i as FAILURE_CLASSES, F as FailureClass, G as GenericSpan, J as JudgeSpan, M as Message, d as RetrievalSpan, g as RunLayer, f as RunStatus, e as SandboxSpan, S as Span, j as SpanBase, c as SpanKind, k as SpanStatus, l as TRACE_SCHEMA_VERSION, T as ToolSpan, a as TraceEvent, m as isJudgeSpan, n as isLlmSpan, o as isRetrievalSpan, p as isSandboxSpan, q as isToolSpan } from './schema-m0gsnbt3.js';
|
|
42
42
|
import { T as TraceStore, R as RunFilter } from './store-CKUAgsJz.js';
|
|
43
43
|
export { E as EventFilter, F as FileSystemTraceStore, a as FileSystemTraceStoreOptions, I as InMemoryTraceStore, S as SpanFilter } from './store-CKUAgsJz.js';
|
|
@@ -56,17 +56,17 @@ import { a as PrmGrader } from './rubric-BOfxn4ja.js';
|
|
|
56
56
|
export { EuRiskClass, GovernanceContext, GovernanceFinding, GovernanceReport, UseCaseSignals, classifyEuAiRisk, euAiActReport, nistAiRmfReport, renderMarkdown, soc2Report, summarize } from './governance/index.js';
|
|
57
57
|
import { b as Layer, S as Severity, L as LayerResult, c as VerifyContext } from './multi-layer-verifier-DlWCXuxL.js';
|
|
58
58
|
export { F as Finding, d as LayerStatus, M as MultiLayerVerifier, a as VerificationReport, V as VerifyOptions, g as gradeSemanticStatus } from './multi-layer-verifier-DlWCXuxL.js';
|
|
59
|
-
import { L as LlmClientOptions } from './llm-client-
|
|
60
|
-
export { d as LlmCallError, b as LlmCallRequest, c as LlmCallResult, e as LlmClient, f as LlmMessage, g as LlmRouteAssertionError, a as LlmRouteRequirements, h as LlmUsage, i as assertLlmRoute, j as backoffMs, k as callLlm, l as callLlmJson, m as isTransientLlmError, p as probeLlm, s as stripFencedJson } from './llm-client-
|
|
61
|
-
export { B as BENCHMARK_SPLIT_SEED, a as BenchmarkAdapter, b as BenchmarkDatasetItem, c as BenchmarkEvaluation, d as benchmarkDeterministicSplit, i as benchmarks } from './index-
|
|
62
|
-
export { C as CallbackResearcher, d as CallbackResearcherOptions, e as CampaignFactoryParams, f as CampaignIntegrityPolicy, g as CampaignRunContext, h as CampaignRunOutcome, i as CampaignRunner, j as CampaignScenario, k as CampaignVariant, c as EvalCampaignOptions, b as EvalCampaignResult, E as ExperimentPlan, a as ExperimentResult, l as FailedRun, F as FailureMode, N as NoopResearcher, R as Researcher, S as SteeringChange, r as runEvalCampaign } from './researcher-
|
|
63
|
-
export { G as GainDistributionBin, a as GainDistributionFigureSpec, b as GainDistributionOptions, m as GateDecision, n as GateEvidence, H as HeldOutGate, o as HeldOutGateConfig, q as HeldOutGateRejectionCode, P as ParetoFigureSpec, c as ParetoPoint, R as RESEARCH_REPORT_HARD_PAIR_FLOOR, d as ResearchReport, e as ResearchReportCandidate, f as ResearchReportDecision, g as ResearchReportMethodology, h as ResearchReportOptions, i as ResearchReportRecommendation, S as SummaryTable, j as SummaryTableOptions, k as SummaryTableRow, l as gainHistogram, p as paretoChart, r as researchReport, s as summaryTable } from './summary-report-
|
|
59
|
+
import { L as LlmClientOptions } from './llm-client-CuUg2Mn3.js';
|
|
60
|
+
export { d as LlmCallError, b as LlmCallRequest, c as LlmCallResult, e as LlmClient, f as LlmMessage, g as LlmRouteAssertionError, a as LlmRouteRequirements, h as LlmUsage, i as assertLlmRoute, j as backoffMs, k as callLlm, l as callLlmJson, m as isTransientLlmError, p as probeLlm, s as stripFencedJson } from './llm-client-CuUg2Mn3.js';
|
|
61
|
+
export { B as BENCHMARK_SPLIT_SEED, a as BenchmarkAdapter, b as BenchmarkDatasetItem, c as BenchmarkEvaluation, d as benchmarkDeterministicSplit, i as benchmarks } from './index-DE3RXAXD.js';
|
|
62
|
+
export { C as CallbackResearcher, d as CallbackResearcherOptions, e as CampaignFactoryParams, f as CampaignIntegrityPolicy, g as CampaignRunContext, h as CampaignRunOutcome, i as CampaignRunner, j as CampaignScenario, k as CampaignVariant, c as EvalCampaignOptions, b as EvalCampaignResult, E as ExperimentPlan, a as ExperimentResult, l as FailedRun, F as FailureMode, N as NoopResearcher, R as Researcher, S as SteeringChange, r as runEvalCampaign } from './researcher-BLPHBbNV.js';
|
|
63
|
+
export { G as GainDistributionBin, a as GainDistributionFigureSpec, b as GainDistributionOptions, m as GateDecision, n as GateEvidence, H as HeldOutGate, o as HeldOutGateConfig, q as HeldOutGateRejectionCode, P as ParetoFigureSpec, c as ParetoPoint, R as RESEARCH_REPORT_HARD_PAIR_FLOOR, d as ResearchReport, e as ResearchReportCandidate, f as ResearchReportDecision, g as ResearchReportMethodology, h as ResearchReportOptions, i as ResearchReportRecommendation, S as SummaryTable, j as SummaryTableOptions, k as SummaryTableRow, l as gainHistogram, p as paretoChart, r as researchReport, s as summaryTable } from './summary-report-Db0dDSWP.js';
|
|
64
64
|
export { I as InterimReleaseConfidence, a as InterimReleaseConfidenceInput, P as PairedEvalueOptions, b as PairedEvalueSequence, c as PairedEvalueStep, S as SequentialDecision, e as evaluateInterimReleaseConfidence, p as pairedEvalueSequence } from './sequential-5iSVfzl2.js';
|
|
65
|
-
import { S as Scenario$1, a as JudgeConfig, G as Gate } from './types-
|
|
66
|
-
import { d as GepaDriverConstraints, R as RunImprovementLoopResult } from './run-improvement-loop-
|
|
65
|
+
import { S as Scenario$1, a as JudgeConfig, G as Gate } from './types-D7lLRYe9.js';
|
|
66
|
+
import { d as GepaDriverConstraints, R as RunImprovementLoopResult } from './run-improvement-loop-D6PZOoQL.js';
|
|
67
67
|
import '@ax-llm/ax';
|
|
68
68
|
import 'zod';
|
|
69
|
-
import './outcome-store-
|
|
69
|
+
import './outcome-store-rnXLEqSn.js';
|
|
70
70
|
|
|
71
71
|
/**
|
|
72
72
|
* Automated pull request opener for the improvement loop.
|
|
@@ -907,271 +907,62 @@ declare class DualAgentBench {
|
|
|
907
907
|
run(config: DualAgentBenchConfig): Promise<DualAgentReport>;
|
|
908
908
|
}
|
|
909
909
|
|
|
910
|
-
interface HostedJudgeDimension {
|
|
911
|
-
name: string;
|
|
912
|
-
weight: number;
|
|
913
|
-
rubric: string;
|
|
914
|
-
}
|
|
915
|
-
interface HostedJudgeConfig {
|
|
916
|
-
model: string;
|
|
917
|
-
mode?: 'llm' | 'sandbox' | 'composite';
|
|
918
|
-
systemPrompt?: string;
|
|
919
|
-
rubricTemplate?: string;
|
|
920
|
-
temperature?: number;
|
|
921
|
-
maxTurns?: number;
|
|
922
|
-
tools?: string[];
|
|
923
|
-
dimensions?: HostedJudgeDimension[];
|
|
924
|
-
setupCommand?: string;
|
|
925
|
-
scripts?: Record<string, string>;
|
|
926
|
-
}
|
|
927
|
-
interface HostedJudgeRequest {
|
|
928
|
-
prompt: string;
|
|
929
|
-
response: string;
|
|
930
|
-
rubric?: string;
|
|
931
|
-
reference?: string;
|
|
932
|
-
judge: HostedJudgeConfig;
|
|
933
|
-
}
|
|
934
|
-
interface HostedJudgeResponse {
|
|
935
|
-
score: number;
|
|
936
|
-
reasoning: string;
|
|
937
|
-
cost: number;
|
|
938
|
-
dimensions?: Array<{
|
|
939
|
-
name: string;
|
|
940
|
-
score: number;
|
|
941
|
-
reasoning: string;
|
|
942
|
-
}>;
|
|
943
|
-
evidence?: Array<{
|
|
944
|
-
type: string;
|
|
945
|
-
content: string;
|
|
946
|
-
}>;
|
|
947
|
-
turns?: number;
|
|
948
|
-
parseFailed?: boolean;
|
|
949
|
-
rawOutput?: string;
|
|
950
|
-
}
|
|
951
|
-
interface HostedRunScoreRequest {
|
|
952
|
-
trace: RunTrace;
|
|
953
|
-
weights?: Partial<RunScoreWeights>;
|
|
954
|
-
driftPatterns?: string[];
|
|
955
|
-
}
|
|
956
|
-
interface HostedRunScoreResponse {
|
|
957
|
-
score: RunScore;
|
|
958
|
-
aggregate: number;
|
|
959
|
-
weights: RunScoreWeights;
|
|
960
|
-
notes: string[];
|
|
961
|
-
}
|
|
962
|
-
type HostedRunCriticConfig = Pick<RunCriticOptions, 'weights'> & {
|
|
963
|
-
driftPatterns?: string[];
|
|
964
|
-
};
|
|
965
|
-
|
|
966
|
-
/**
|
|
967
|
-
* Experiment tracker — group runs, diff them, watch scores move over time.
|
|
968
|
-
*
|
|
969
|
-
* Not MLflow. Not Weights & Biases. Just the 20% that actually ships:
|
|
970
|
-
* - A run has a config (prompt hash, model, scenario ids, seed)
|
|
971
|
-
* - Runs belong to experiments (named groups)
|
|
972
|
-
* - The store is pluggable (in-memory for tests, filesystem for local,
|
|
973
|
-
* custom for Langfuse/D1)
|
|
974
|
-
* - Diffs show score deltas, new/dropped scenarios, and config changes
|
|
975
|
-
*
|
|
976
|
-
* The output plugs directly into `BenchmarkReport` — runs archive the full
|
|
977
|
-
* report, diff operates on the summary.
|
|
978
|
-
*/
|
|
979
|
-
|
|
980
|
-
interface RunConfig {
|
|
981
|
-
experimentId: string;
|
|
982
|
-
name?: string;
|
|
983
|
-
model?: string;
|
|
984
|
-
promptHash?: string;
|
|
985
|
-
promptVersion?: string;
|
|
986
|
-
seed?: number;
|
|
987
|
-
metadata?: Record<string, unknown>;
|
|
988
|
-
}
|
|
989
|
-
interface Run {
|
|
990
|
-
id: string;
|
|
991
|
-
experimentId: string;
|
|
992
|
-
name?: string;
|
|
993
|
-
config: RunConfig;
|
|
994
|
-
startedAt: string;
|
|
995
|
-
completedAt?: string;
|
|
996
|
-
status: 'running' | 'completed' | 'failed';
|
|
997
|
-
report?: BenchmarkReport;
|
|
998
|
-
error?: string;
|
|
999
|
-
}
|
|
1000
|
-
interface Experiment {
|
|
1001
|
-
id: string;
|
|
1002
|
-
name: string;
|
|
1003
|
-
createdAt: string;
|
|
1004
|
-
metadata?: Record<string, unknown>;
|
|
1005
|
-
}
|
|
1006
|
-
interface ExperimentStore {
|
|
1007
|
-
saveExperiment(exp: Experiment): Promise<void>;
|
|
1008
|
-
getExperiment(id: string): Promise<Experiment | null>;
|
|
1009
|
-
listExperiments(): Promise<Experiment[]>;
|
|
1010
|
-
saveRun(run: Run): Promise<void>;
|
|
1011
|
-
getRun(id: string): Promise<Run | null>;
|
|
1012
|
-
listRuns(experimentId: string): Promise<Run[]>;
|
|
1013
|
-
}
|
|
1014
|
-
declare class InMemoryExperimentStore implements ExperimentStore {
|
|
1015
|
-
private readonly experiments;
|
|
1016
|
-
private readonly runs;
|
|
1017
|
-
saveExperiment(exp: Experiment): Promise<void>;
|
|
1018
|
-
getExperiment(id: string): Promise<Experiment | null>;
|
|
1019
|
-
listExperiments(): Promise<Experiment[]>;
|
|
1020
|
-
saveRun(run: Run): Promise<void>;
|
|
1021
|
-
getRun(id: string): Promise<Run | null>;
|
|
1022
|
-
listRuns(experimentId: string): Promise<Run[]>;
|
|
1023
|
-
}
|
|
1024
|
-
declare class ExperimentTracker {
|
|
1025
|
-
private readonly store;
|
|
1026
|
-
constructor(store: ExperimentStore);
|
|
1027
|
-
startExperiment(name: string, metadata?: Record<string, unknown>): Promise<Experiment>;
|
|
1028
|
-
startRun(config: RunConfig): Promise<Run>;
|
|
1029
|
-
completeRun(runId: string, report: BenchmarkReport): Promise<void>;
|
|
1030
|
-
failRun(runId: string, error: string): Promise<void>;
|
|
1031
|
-
/**
|
|
1032
|
-
* Diff two completed runs. Returns per-scenario deltas, aggregate delta,
|
|
1033
|
-
* and config changes that may explain the movement.
|
|
1034
|
-
*/
|
|
1035
|
-
diff(runIdA: string, runIdB: string): Promise<RunDiff>;
|
|
1036
|
-
/** Timeline of aggregate scores for an experiment. */
|
|
1037
|
-
timeline(experimentId: string): Promise<Array<{
|
|
1038
|
-
runId: string;
|
|
1039
|
-
startedAt: string;
|
|
1040
|
-
overall: number | null;
|
|
1041
|
-
}>>;
|
|
1042
|
-
}
|
|
1043
|
-
interface RunDiff {
|
|
1044
|
-
before: {
|
|
1045
|
-
runId: string;
|
|
1046
|
-
name?: string;
|
|
1047
|
-
startedAt: string;
|
|
1048
|
-
};
|
|
1049
|
-
after: {
|
|
1050
|
-
runId: string;
|
|
1051
|
-
name?: string;
|
|
1052
|
-
startedAt: string;
|
|
1053
|
-
};
|
|
1054
|
-
aggregateDelta: number;
|
|
1055
|
-
scenarios: Array<{
|
|
1056
|
-
scenarioId: string;
|
|
1057
|
-
before: number | null;
|
|
1058
|
-
after: number | null;
|
|
1059
|
-
delta: number | null;
|
|
1060
|
-
status: 'improved' | 'regressed' | 'unchanged' | 'added' | 'removed';
|
|
1061
|
-
}>;
|
|
1062
|
-
configChanges: Record<string, {
|
|
1063
|
-
before: unknown;
|
|
1064
|
-
after: unknown;
|
|
1065
|
-
}>;
|
|
1066
|
-
}
|
|
1067
|
-
|
|
1068
910
|
/**
|
|
1069
|
-
*
|
|
1070
|
-
*
|
|
1071
|
-
* Workers-safe (uses only the `D1Database` binding the runtime injects). Two
|
|
1072
|
-
* tables, no joins, no migrations beyond `ensureSchema()`. Schema designed so
|
|
1073
|
-
* a Worker route can both write the row at run start and update it at run end
|
|
1074
|
-
* without losing the original config — the row's lifecycle mirrors the
|
|
1075
|
-
* `Run.status` field one-to-one.
|
|
911
|
+
* Judge-ensemble reducer — folds N independent judge verdicts on the same
|
|
912
|
+
* artifact into one aggregate score.
|
|
1076
913
|
*
|
|
1077
|
-
*
|
|
1078
|
-
*
|
|
1079
|
-
*
|
|
1080
|
-
*
|
|
1081
|
-
*
|
|
914
|
+
* The pattern every multi-model judge re-implements: run K uncorrelated judges
|
|
915
|
+
* (different model families remove within-family bias), then reduce their
|
|
916
|
+
* per-dimension verdicts to a single composite + an inter-rater disagreement
|
|
917
|
+
* signal. This is the pure reduction — no LLM, no I/O — so a lift it produces
|
|
918
|
+
* is attributable to the scores, not to where they came from.
|
|
1082
919
|
*
|
|
1083
|
-
*
|
|
1084
|
-
*
|
|
1085
|
-
*
|
|
920
|
+
* Fail-loud: a judge that errored or returned malformed output is recorded in
|
|
921
|
+
* `failedJudges` with `perDimension: null`, never folded into a zero. If EVERY
|
|
922
|
+
* judge failed the reducer throws — a silent zero here would corrupt the gate's
|
|
923
|
+
* number. A failed judge still burned tokens, so its `costUsd` is still summed.
|
|
1086
924
|
*/
|
|
1087
|
-
|
|
1088
|
-
|
|
1089
|
-
|
|
1090
|
-
|
|
1091
|
-
|
|
1092
|
-
*/
|
|
1093
|
-
|
|
1094
|
-
|
|
1095
|
-
|
|
1096
|
-
|
|
1097
|
-
|
|
1098
|
-
|
|
1099
|
-
|
|
1100
|
-
|
|
1101
|
-
|
|
1102
|
-
|
|
1103
|
-
|
|
1104
|
-
|
|
1105
|
-
|
|
1106
|
-
|
|
1107
|
-
/**
|
|
1108
|
-
|
|
1109
|
-
/**
|
|
1110
|
-
|
|
1111
|
-
|
|
1112
|
-
|
|
1113
|
-
|
|
1114
|
-
|
|
1115
|
-
}
|
|
1116
|
-
declare class D1ExperimentStore implements ExperimentStore {
|
|
1117
|
-
private readonly db;
|
|
1118
|
-
private readonly experimentsTable;
|
|
1119
|
-
private readonly runsTable;
|
|
1120
|
-
private readonly metaTable;
|
|
1121
|
-
private schemaReady;
|
|
1122
|
-
constructor(options: D1ExperimentStoreOptions);
|
|
1123
|
-
/**
|
|
1124
|
-
* Idempotent schema setup. Safe to call before every operation; the second
|
|
1125
|
-
* call short-circuits via `schemaReady`. Most consumers will call it once
|
|
1126
|
-
* during Worker bootstrap.
|
|
1127
|
-
*/
|
|
1128
|
-
ensureSchema(): Promise<void>;
|
|
1129
|
-
saveExperiment(exp: Experiment): Promise<void>;
|
|
1130
|
-
getExperiment(id: string): Promise<Experiment | null>;
|
|
1131
|
-
listExperiments(): Promise<Experiment[]>;
|
|
1132
|
-
saveRun(run: Run): Promise<void>;
|
|
1133
|
-
getRun(id: string): Promise<Run | null>;
|
|
1134
|
-
listRuns(experimentId: string): Promise<Run[]>;
|
|
925
|
+
/** One judge's verdict. `perDimension: null` ⇒ that judge call failed (threw or
|
|
926
|
+
* returned malformed output) — recorded, never folded into a zero. */
|
|
927
|
+
interface JudgeVerdict<D extends string = string> {
|
|
928
|
+
/** Judge identity (typically the model id) — the `perJudge` / `failedJudges` key. */
|
|
929
|
+
model: string;
|
|
930
|
+
/** Per-dimension scores, or `null` when the judge failed. */
|
|
931
|
+
perDimension: Record<D, number> | null;
|
|
932
|
+
/** Optional one-line rationale; the first non-empty one becomes the aggregate's. */
|
|
933
|
+
rationale?: string;
|
|
934
|
+
/** Optional reported cost — summed across ALL verdicts (failed included). */
|
|
935
|
+
costUsd?: number;
|
|
936
|
+
}
|
|
937
|
+
/** The aggregated ensemble result. */
|
|
938
|
+
interface EnsembleAggregate<D extends string = string> {
|
|
939
|
+
/** Mean over SURVIVING judges, per dimension (absent-everywhere ⇒ 0). */
|
|
940
|
+
perDimension: Record<D, number>;
|
|
941
|
+
/** Weighted mean of `perDimension` (uniform unless `weights` given). */
|
|
942
|
+
composite: number;
|
|
943
|
+
/** Each surviving judge's clamped per-dimension scores, for trace drill-down. */
|
|
944
|
+
perJudge: Record<string, Record<D, number>>;
|
|
945
|
+
/** Max over dimensions of (max − min) across survivors — the inter-rater signal. */
|
|
946
|
+
maxDisagreement: number;
|
|
947
|
+
/** Models whose verdict was null (failed). */
|
|
948
|
+
failedJudges: string[];
|
|
949
|
+
/** Sum of `costUsd` over ALL verdicts, failed included. */
|
|
950
|
+
costUsd: number;
|
|
951
|
+
/** First non-empty survivor rationale, or `'llm-judge'`. */
|
|
952
|
+
rationale: string;
|
|
1135
953
|
}
|
|
1136
|
-
|
|
1137
954
|
/**
|
|
1138
|
-
*
|
|
955
|
+
* Reduce per-judge verdicts to one aggregate. Generic over the rubric: pass the
|
|
956
|
+
* stable-ordered `dimensionKeys` the judges scored.
|
|
1139
957
|
*
|
|
1140
|
-
*
|
|
1141
|
-
*
|
|
1142
|
-
*
|
|
1143
|
-
*
|
|
1144
|
-
*
|
|
1145
|
-
*
|
|
1146
|
-
* archives), latest-write-wins per `id`. Subsequent writes update the
|
|
1147
|
-
* in-memory index in place so reads after writes are O(1).
|
|
1148
|
-
*
|
|
1149
|
-
* Node-only — imports `node:fs/promises`. Don't import this from a Worker;
|
|
1150
|
-
* use the in-memory store or the D1 store from `./experiment-tracker-d1`.
|
|
958
|
+
* - Per-dimension score = mean over surviving judges (out-of-range clamped to [0,1]).
|
|
959
|
+
* - `composite` = `weightedComposite` of `perDimension`. No `weights` ⇒ uniform
|
|
960
|
+
* over every dimension. A partial `weights` map selects AND weights exactly the
|
|
961
|
+
* named dimensions (others excluded from the composite) — the substrate's
|
|
962
|
+
* sum-normalized weighting, not re-implemented here.
|
|
963
|
+
* - Throws if `verdicts` or `dimensionKeys` is empty, or if every judge failed.
|
|
1151
964
|
*/
|
|
1152
|
-
|
|
1153
|
-
interface FileSystemExperimentStoreOptions {
|
|
1154
|
-
/** Directory the NDJSON files live in. Created on first write. */
|
|
1155
|
-
dir: string;
|
|
1156
|
-
/** Bytes after which a file is rolled over. Default 32 MB (matches FileSystemTraceStore). */
|
|
1157
|
-
maxBytes?: number;
|
|
1158
|
-
}
|
|
1159
|
-
declare class FileSystemExperimentStore implements ExperimentStore {
|
|
1160
|
-
private readonly dir;
|
|
1161
|
-
private readonly maxBytes;
|
|
1162
|
-
private index?;
|
|
1163
|
-
private loaded;
|
|
1164
|
-
constructor(options: FileSystemExperimentStoreOptions);
|
|
1165
|
-
saveExperiment(exp: Experiment): Promise<void>;
|
|
1166
|
-
getExperiment(id: string): Promise<Experiment | null>;
|
|
1167
|
-
listExperiments(): Promise<Experiment[]>;
|
|
1168
|
-
saveRun(run: Run): Promise<void>;
|
|
1169
|
-
getRun(id: string): Promise<Run | null>;
|
|
1170
|
-
listRuns(experimentId: string): Promise<Run[]>;
|
|
1171
|
-
private ensureDir;
|
|
1172
|
-
private append;
|
|
1173
|
-
private load;
|
|
1174
|
-
}
|
|
965
|
+
declare function aggregateJudgeVerdicts<D extends string>(verdicts: readonly JudgeVerdict<D>[], dimensionKeys: readonly D[], weights?: Partial<Record<D, number>>): EnsembleAggregate<D>;
|
|
1175
966
|
|
|
1176
967
|
type SandboxJudgeKind = 'compiler' | 'test' | 'linter' | 'security';
|
|
1177
968
|
interface SandboxJudgeSpec {
|
|
@@ -2096,7 +1887,7 @@ interface ContractMetric {
|
|
|
2096
1887
|
/** Max tolerated regression (e.g. 0.02 = 2pp worse than baseline). */
|
|
2097
1888
|
maxRegression?: number;
|
|
2098
1889
|
/** Optional extractor if the metric isn't in the default set. */
|
|
2099
|
-
extract?: (run: Run
|
|
1890
|
+
extract?: (run: Run, store: TraceStore) => Promise<number | null>;
|
|
2100
1891
|
}
|
|
2101
1892
|
interface ThresholdContract {
|
|
2102
1893
|
name: string;
|
|
@@ -3787,7 +3578,7 @@ interface ReviewerSoftFailDefaults {
|
|
|
3787
3578
|
interface CreateDefaultReviewerOptions {
|
|
3788
3579
|
/** Model id to call. */
|
|
3789
3580
|
model: string;
|
|
3790
|
-
/** Per-call timeout. Default
|
|
3581
|
+
/** Per-call timeout. Default 300s. */
|
|
3791
3582
|
timeoutMs?: number;
|
|
3792
3583
|
/** LlmClient transport config (baseUrl, apiKey, authHeader, etc.). */
|
|
3793
3584
|
llm?: LlmClientOptions;
|
|
@@ -4068,7 +3859,7 @@ declare function precision<T>(goldens: GoldenSpec[], candidates: T[], options?:
|
|
|
4068
3859
|
interface JudgeRetryPolicy {
|
|
4069
3860
|
/** Max attempts per model. Default 3 (one initial + two retries). */
|
|
4070
3861
|
maxAttempts?: number;
|
|
4071
|
-
/** Per-attempt timeout in ms. Default
|
|
3862
|
+
/** Per-attempt timeout in ms. Default 300_000. */
|
|
4072
3863
|
timeoutMs?: number;
|
|
4073
3864
|
/**
|
|
4074
3865
|
* Models to try, in order. The first model is the primary; subsequent
|
|
@@ -4819,4 +4610,4 @@ declare namespace index {
|
|
|
4819
4610
|
export { type index_AgentProfile as AgentProfile, type index_AgentProfileSection as AgentProfileSection, index_BASELINE_ROLES as BASELINE_ROLES, type index_BaselineRoleKey as BaselineRoleKey, type index_ProfileSkill as ProfileSkill, index_applyDomainPatch as applyDomainPatch, index_baselineProfile as baselineProfile, index_baselineProfileFromRole as baselineProfileFromRole, index_engineerRole as engineerRole, index_generalistRole as generalistRole, index_prodProfile as prodProfile, index_profileToSurface as profileToSurface, index_renderProfile as renderProfile, index_researcherRole as researcherRole, index_sectionHash as sectionHash };
|
|
4820
4611
|
}
|
|
4821
4612
|
|
|
4822
|
-
export { type ActiveLearningOptions, type AdapterRun, AgentDriver, type AgentDriverConfig, AgentEvalError, AgentProfile$1 as AgentProfile, type AgreementResult, type AlignmentOp, AnalyzeTracesInput, AnalyzeTracesOptions, AnalyzeTracesResult, type AntiSlopConfig, type AntiSlopIssue, type AntiSlopReport, type AssertCrossFamilyOptions, type AssertSingleBackendOptions, type AutoPrClient, AxGepaSteeringOptimizer, type AxSteeringOptimizerConfig, type BackendDescriptor, BaselineReport, BehaviorAssertion, BenchmarkReport, BenchmarkRunner, BenchmarkRunnerConfig, type BisectOptions, type BisectResult, type BisectStep, BudgetBreachError, BudgetGuard, BudgetLedgerEntry, BudgetSpec, type BuildAgreementJudgeOptions, CallExpectation, type CanaryAlert, type CanaryKind, type CanaryLeak, type CanaryOptions, type CanaryReport, type CanarySeverity, type CandidateScenario, type CausalAttributionReport, type CellVerdict, ChatRequest, CheckResult, CollectedArtifacts, type CommandRunner, type CompareLabels, CompletionCriterion, type ContinuityCheck, type ContinuityCheckResult, type ContinuityReport, type ContinuitySnapshotPair, type ContractMetric, type ContractReport, ConvergenceTracker, type CostEntry, type CostSummary, CostTracker, type CounterfactualContext, type CounterfactualMutation, type CounterfactualResult, type CounterfactualRunner, CreateChatClientOpts, type CreateDefaultReviewerOptions, type CreateSandboxPoolOpts, CrossFamilyError, type CrossTraceDiff, type CrossTraceDiffOptions, D1ExperimentStore, type D1ExperimentStoreOptions, type D1Like, type D1PreparedStatementLike, DEFAULT_AGENT_SLOS, DEFAULT_FINDERS, DEFAULT_MUTATION_PRIMITIVES, DEFAULT_MUTATORS, DEFAULT_PR_REVIEW_SCORE_WEIGHTS, DEFAULT_SEVERITY_WEIGHTS, Dataset, DatasetScenario, type DecideNextUserTurnOpts, type DeployFamily, type DeployGateLayerInput, type DeployRunResult, type DeployRunner, type DiffScorecardOptions, type DirEntry, type DiscoverPersonasOptions, type DiscoveredPersona, DriverResult, DriverState, DualAgentBench, type DualAgentBenchConfig, type DualAgentReport, type DualAgentRound, type DualAgentScenario, type DualAgentScenarioResult, ERROR_COUNT_PATTERNS, type ErrorCountPattern, type EvolutionRound, type ExecutorConfig, type Expectation, type Experiment, type Run as ExperimentRun, type ExperimentStore, ExperimentTracker, type ExportedRewardModel, type ExtractOptions, type ExtractResult, type FactorContribution, type FactorialCell, FeedbackLabel, FeedbackTrajectory, FeedbackTrajectoryStore, type FieldAgreementSpec, type FileChange, FileSystemExperimentStore, type FileSystemExperimentStoreOptions, type FlowAction, type FlowLayerEnv, type FlowLayerFactoryInput, type FlowRunner, type FlowRunnerStepResult, type FlowSpec, type FlowStep, type GhCliClientOptions, type GoldScenario, type GoldSplit, type GoldenSeverity, type GoldenSpec, HarnessConfig, HoldoutAuditor, type HostedJudgeConfig, type HostedJudgeDimension, type HostedJudgeRequest, type HostedJudgeResponse, type HostedRunCriticConfig, type HostedRunScoreRequest, type HostedRunScoreResponse, type HttpGithubClientOptions, type HypothesisManifest, type HypothesisResult, INTENT_MATCH_JUDGE_VERSION, type ImageData, InMemoryExperimentStore, InMemoryWorkspaceInspector, type InferenceScorer, type InspectorContext, type IntentMatchInput, type IntentMatchOptions, type IntentMatchResult, type InteractionContribution, type JudgeFamily, type JudgeFleetOptions, JudgeFn, type JudgeReplayResult, type JudgeRetryOutcome, type JudgeRetryPolicy, JudgeRunner, type KeywordConceptSpec, type KeywordCoverageFinding, type KeywordCoverageOptions, type KeywordCoverageResult, type LangfuseEnvelope, type LangfuseGeneration, type LangfuseScore, Layer, LayerResult, type LiveProofArtifact, type LiveProofConfig, type LiveProofContext, type LiveProofResult, LlmClientOptions, LlmSpan, LockedJsonlAppender, MODEL_PRICING, type MatchResult, type MatcherResult, type MergeOptions, MetricsCollector, type MuffledFinder, type MuffledFinding, type MultiToolchainLayerConfig, type Mutator, Mutex, type Oracle, type OracleObservation, type OracleReport, type OracleResult, type OrthogonalityInput, type OrthogonalityResult, OtelExportConfig, OtelExporter, type OtelPipelineHandle, type OtelPipelineOptions, PairwiseSteeringOptimizer, type ParaphraseRobustnessScenarioInput, type ParaphraseRobustnessScenarioResult, type ParseStudentLabel, PersonaConfig, type Playbook, type PlaybookEntry, type PoolSlot, type PrReviewAuditCase, type PrReviewBenchmarkSummary, type PrReviewComment, type PrReviewMatchedFinding, type PrReviewOutcome, type PrReviewReferenceFinding, type PrReviewScore, type PrReviewScoreWeights, type PrReviewSeverity, type PrReviewSource, ProductClient, ProductClientConfig, type PromptHandle, PromptRegistry, type ProposeAutomatedPullRequestInput, type ProposeAutomatedPullRequestResult, type RecordRunsOptions, type ReferenceMatchResult, type ReferenceReplayAdapter, type ReferenceReplayAdapterFn, type ReferenceReplayAdapterLike, type ReferenceReplayAggregate, type ReferenceReplayCandidate, type ReferenceReplayCase, type ReferenceReplayCaseRun, type ReferenceReplayExecutionScenario, type ReferenceReplayItem, type ReferenceReplayMatch, type ReferenceReplayMatchStrategy, type ReferenceReplayMatcher, type ReferenceReplayPromotionDecision, type ReferenceReplayPromotionPolicy, type ReferenceReplayRun, type ReferenceReplayRunContext, type ReferenceReplayRunOptions, type ReferenceReplayRunStore, type ReferenceReplayScenario, type ReferenceReplayScenarioScore, type ReferenceReplayScore, type ReferenceReplayScoreOptions, type ReferenceReplaySplit, type ReferenceReplaySplitComparison, type ReferenceReplaySteeringRowsOptions, type ReflectionContext, type ReflectionProposal, ReleaseConfidenceScorecard, ReleaseConfidenceThresholds, type RenderStudentPrompt, type RepoRef, type ReviewerMemoryEntry, type ReviewerOutput, type ReviewerPromptInput, type ReviewerSoftFailDefaults, type ReviewerVerificationSummary, type RobustnessResult, Run$1 as Run, type RunCommandInput, type RunCommandResult, type RunConfig, RunCriticOptions, type RunDiff, type RunDistillationOptions, type RunDistillationResult, RunFilter, RunRecord, RunScore, RunScoreWeights, RunTrace, SandboxDriver, SandboxHarnessResult, type SandboxJudgeKind, type SandboxJudgeResult, type SandboxJudgeSpec, type SandboxPool, type ScanOptions, Scenario, type ScenarioCost, ScenarioFile, ScenarioRegistry, ScenarioResult, type Scorecard, type ScorecardCell, type ScorecardCellDiff, type ScorecardDiff, type ScorecardEntry, type ScorecardLogLine, type ScoredTarget, type SelfPlayOptions, type SelfPlayProposer, type SelfPlayScorer, type SeriesConvergenceOptions, type SeriesConvergenceResult, Severity, type SignedManifest, type SignedManifestAlgo, type SingleBackendDivergence, SingleBackendError, type SingleBackendField, type SingleBackendReport, type Slo, type SloCheckResult, type SloComparator, type SloReport, type SloSeverity, type SlopCategory, type SlotFactory, type SplitGoldOptions, SteeringBundle, type SteeringOptimizationResult, type SteeringOptimizationRow, type SteeringOptimizationSelector, type SteeringOptimizerBackend, type SteeringOptimizerConfig, type StepAttribution, type SynthesisReason, type SynthesisTarget, TestResult, type ThresholdContract, TokenCounter, type TokenSpec, TraceEmitter, TraceStore, type TracedAnalystOptions, type TracedJudgeOptions, Trajectory, TrajectoryStep, type TrialTrace, TurnMetrics, UI_FINDING_SEVERITIES, UI_LENSES, UNIVERSAL_FINDERS, type UiFinding, type UiFindingScreenshot, type UiFindingSeverity, type UiLens, VerifyContext, type VisualDiffOptions, type VisualDiffResult, type ViteDeployRunnerInput, type WorkspaceAssertion, type WorkspaceAssertionResult, type WorkspaceInspector, type WorkspaceSnapshot, type WranglerDeployRunnerInput, adversarialJudge, aggregatePrReviewScore, analyzeAntiSlop, analyzeSeries, appendScorecard, assertCrossFamily, assertSingleBackend, attributeCounterfactuals, bisect, buildAgreementJudge, buildDriverSystemPrompt, buildReflectionPrompt, buildReviewerPrompt, canaryLeakView, canonicalize, causalAttribution, checkBehavioralCanary, checkCanaries, checkSlos, codeExecutionJudge, coherenceJudge, collectionPreserved, commentsForSource, commitBisect, compareReferenceReplay, compilerJudge, createAntiSlopJudge, createCustomJudge, createDefaultReviewer, createDomainExpertJudge, createIntentMatchJudge, createSandboxPool, crossTraceDiff, decideNextUserTurn, decideReferenceReplayPromotion, decideReferenceReplayRunPromotion, defaultJudges, defaultParseStudentLabel, defaultReferenceReplayMatcher, defaultRenderStudentPrompt, deployGateLayer, diffScorecard, discoverPersonas, distillPlaybook, estimateCost, estimateTokens, evaluateContract, evaluateHypothesis, evaluateOracles, executeScenario, expectAgent, exportRewardModel, extractAssetUrls, extractErrorCount, fieldAgreement, fileContains, fileExists, findAutoMatchNoExpectation, findConstructorCwdDropped, findFallbackToPass, findLiteralTruePass, findSkipCountsAsPass, flowLayer, formatBenchmarkReport, formatDriverReport, formatFindings, formatScorecardDiff, ghCliClient, precision as goldenPrecision, hashContent, hashJson, htmlContainsElement, httpGithubClient, inMemoryReferenceReplayStore, isModelPriced, isOtelConfigured, jsonShape, jsonlReferenceReplayStore, judgeFamily, keyPreserved, linterJudge, loadGoldScenarios, loadScorecard, loadScorerFromGrader, localCommandRunner, lowercaseMutator, matchGoldens, mergeLayerResults, multiToolchainLayer, notBlocked, paraphraseRobustness, paraphraseRobustnessScenarios, parseGoldJsonl, parseReflectionResponse, passOrthogonality, pixelDeltaRatio, politenessPrefixMutator, printDriverSummary, index as profile, promptBisect, proposeAutomatedPullRequest, proposeSynthesisTargets, recordRuns, recordRunsToScorecard, referenceReplayRunsToSteeringRows, referenceReplayScenarioToRunScore, regexMatches, renderMarkdownReport, renderPlaybookMarkdown, replayScorerOverCorpus, replayTraceThroughJudge, resetLockedAppendersForTesting, resolveModelPricing, rowCount, rowWhere, runAssertions, runBehavioralCanaries, runCanaries, runCounterfactual, runDistillation, runE2EWorkflow, runExpectations, runIntentMatchJudge, runJudgeFleet, runKeywordCoverageJudge, runKeywordCoverageJudgeUrl, runLiveProof, runReferenceReplay, runSelfPlay, scanForMuffledGates, scoreContinuity, scorePrReviewComments, scorePrReviewSource, scoreReferenceReplay, securityJudge, sentenceReorderMutator, signManifest, splitGold, statusAdvanced, summarizePrReviewBenchmark, testJudge, textInSnapshot, toLangfuseEnvelope, toPrometheusText, traceJudge, traceJudgeEnsemble, tracedAnalyzeTraces, typoMutator, urlContains, verifyManifest, visualDiff, viteDeployRunner, weightedRecall, whitespaceCollapseMutator, withJudgeRetry, withOtelPipeline, wranglerDeployRunner };
|
|
4613
|
+
export { type ActiveLearningOptions, type AdapterRun, AgentDriver, type AgentDriverConfig, AgentEvalError, AgentProfile$1 as AgentProfile, type AgreementResult, type AlignmentOp, AnalyzeTracesInput, AnalyzeTracesOptions, AnalyzeTracesResult, type AntiSlopConfig, type AntiSlopIssue, type AntiSlopReport, type AssertCrossFamilyOptions, type AssertSingleBackendOptions, type AutoPrClient, AxGepaSteeringOptimizer, type AxSteeringOptimizerConfig, type BackendDescriptor, BaselineReport, BehaviorAssertion, BenchmarkReport, BenchmarkRunner, BenchmarkRunnerConfig, type BisectOptions, type BisectResult, type BisectStep, BudgetBreachError, BudgetGuard, BudgetLedgerEntry, BudgetSpec, type BuildAgreementJudgeOptions, CallExpectation, type CanaryAlert, type CanaryKind, type CanaryLeak, type CanaryOptions, type CanaryReport, type CanarySeverity, type CandidateScenario, type CausalAttributionReport, type CellVerdict, ChatRequest, CheckResult, CollectedArtifacts, type CommandRunner, type CompareLabels, CompletionCriterion, type ContinuityCheck, type ContinuityCheckResult, type ContinuityReport, type ContinuitySnapshotPair, type ContractMetric, type ContractReport, ConvergenceTracker, type CostEntry, type CostSummary, CostTracker, type CounterfactualContext, type CounterfactualMutation, type CounterfactualResult, type CounterfactualRunner, CreateChatClientOpts, type CreateDefaultReviewerOptions, type CreateSandboxPoolOpts, CrossFamilyError, type CrossTraceDiff, type CrossTraceDiffOptions, DEFAULT_AGENT_SLOS, DEFAULT_FINDERS, DEFAULT_MUTATION_PRIMITIVES, DEFAULT_MUTATORS, DEFAULT_PR_REVIEW_SCORE_WEIGHTS, DEFAULT_SEVERITY_WEIGHTS, Dataset, DatasetScenario, type DecideNextUserTurnOpts, type DeployFamily, type DeployGateLayerInput, type DeployRunResult, type DeployRunner, type DiffScorecardOptions, type DirEntry, type DiscoverPersonasOptions, type DiscoveredPersona, DriverResult, DriverState, DualAgentBench, type DualAgentBenchConfig, type DualAgentReport, type DualAgentRound, type DualAgentScenario, type DualAgentScenarioResult, ERROR_COUNT_PATTERNS, type EnsembleAggregate, type ErrorCountPattern, type EvolutionRound, type ExecutorConfig, type Expectation, type ExportedRewardModel, type ExtractOptions, type ExtractResult, type FactorContribution, type FactorialCell, FeedbackLabel, FeedbackTrajectory, FeedbackTrajectoryStore, type FieldAgreementSpec, type FileChange, type FlowAction, type FlowLayerEnv, type FlowLayerFactoryInput, type FlowRunner, type FlowRunnerStepResult, type FlowSpec, type FlowStep, type GhCliClientOptions, type GoldScenario, type GoldSplit, type GoldenSeverity, type GoldenSpec, HarnessConfig, HoldoutAuditor, type HttpGithubClientOptions, type HypothesisManifest, type HypothesisResult, INTENT_MATCH_JUDGE_VERSION, type ImageData, InMemoryWorkspaceInspector, type InferenceScorer, type InspectorContext, type IntentMatchInput, type IntentMatchOptions, type IntentMatchResult, type InteractionContribution, type JudgeFamily, type JudgeFleetOptions, JudgeFn, type JudgeReplayResult, type JudgeRetryOutcome, type JudgeRetryPolicy, JudgeRunner, type JudgeVerdict, type KeywordConceptSpec, type KeywordCoverageFinding, type KeywordCoverageOptions, type KeywordCoverageResult, type LangfuseEnvelope, type LangfuseGeneration, type LangfuseScore, Layer, LayerResult, type LiveProofArtifact, type LiveProofConfig, type LiveProofContext, type LiveProofResult, LlmClientOptions, LlmSpan, LockedJsonlAppender, MODEL_PRICING, type MatchResult, type MatcherResult, type MergeOptions, MetricsCollector, type MuffledFinder, type MuffledFinding, type MultiToolchainLayerConfig, type Mutator, Mutex, type Oracle, type OracleObservation, type OracleReport, type OracleResult, type OrthogonalityInput, type OrthogonalityResult, OtelExportConfig, OtelExporter, type OtelPipelineHandle, type OtelPipelineOptions, PairwiseSteeringOptimizer, type ParaphraseRobustnessScenarioInput, type ParaphraseRobustnessScenarioResult, type ParseStudentLabel, PersonaConfig, type Playbook, type PlaybookEntry, type PoolSlot, type PrReviewAuditCase, type PrReviewBenchmarkSummary, type PrReviewComment, type PrReviewMatchedFinding, type PrReviewOutcome, type PrReviewReferenceFinding, type PrReviewScore, type PrReviewScoreWeights, type PrReviewSeverity, type PrReviewSource, ProductClient, ProductClientConfig, type PromptHandle, PromptRegistry, type ProposeAutomatedPullRequestInput, type ProposeAutomatedPullRequestResult, type RecordRunsOptions, type ReferenceMatchResult, type ReferenceReplayAdapter, type ReferenceReplayAdapterFn, type ReferenceReplayAdapterLike, type ReferenceReplayAggregate, type ReferenceReplayCandidate, type ReferenceReplayCase, type ReferenceReplayCaseRun, type ReferenceReplayExecutionScenario, type ReferenceReplayItem, type ReferenceReplayMatch, type ReferenceReplayMatchStrategy, type ReferenceReplayMatcher, type ReferenceReplayPromotionDecision, type ReferenceReplayPromotionPolicy, type ReferenceReplayRun, type ReferenceReplayRunContext, type ReferenceReplayRunOptions, type ReferenceReplayRunStore, type ReferenceReplayScenario, type ReferenceReplayScenarioScore, type ReferenceReplayScore, type ReferenceReplayScoreOptions, type ReferenceReplaySplit, type ReferenceReplaySplitComparison, type ReferenceReplaySteeringRowsOptions, type ReflectionContext, type ReflectionProposal, ReleaseConfidenceScorecard, ReleaseConfidenceThresholds, type RenderStudentPrompt, type RepoRef, type ReviewerMemoryEntry, type ReviewerOutput, type ReviewerPromptInput, type ReviewerSoftFailDefaults, type ReviewerVerificationSummary, type RobustnessResult, Run, type RunCommandInput, type RunCommandResult, type RunDistillationOptions, type RunDistillationResult, RunFilter, RunRecord, RunScore, RunScoreWeights, SandboxDriver, SandboxHarnessResult, type SandboxJudgeKind, type SandboxJudgeResult, type SandboxJudgeSpec, type SandboxPool, type ScanOptions, Scenario, type ScenarioCost, ScenarioFile, ScenarioRegistry, ScenarioResult, type Scorecard, type ScorecardCell, type ScorecardCellDiff, type ScorecardDiff, type ScorecardEntry, type ScorecardLogLine, type ScoredTarget, type SelfPlayOptions, type SelfPlayProposer, type SelfPlayScorer, type SeriesConvergenceOptions, type SeriesConvergenceResult, Severity, type SignedManifest, type SignedManifestAlgo, type SingleBackendDivergence, SingleBackendError, type SingleBackendField, type SingleBackendReport, type Slo, type SloCheckResult, type SloComparator, type SloReport, type SloSeverity, type SlopCategory, type SlotFactory, type SplitGoldOptions, SteeringBundle, type SteeringOptimizationResult, type SteeringOptimizationRow, type SteeringOptimizationSelector, type SteeringOptimizerBackend, type SteeringOptimizerConfig, type StepAttribution, type SynthesisReason, type SynthesisTarget, TestResult, type ThresholdContract, TokenCounter, type TokenSpec, TraceEmitter, TraceStore, type TracedAnalystOptions, type TracedJudgeOptions, Trajectory, TrajectoryStep, type TrialTrace, TurnMetrics, UI_FINDING_SEVERITIES, UI_LENSES, UNIVERSAL_FINDERS, type UiFinding, type UiFindingScreenshot, type UiFindingSeverity, type UiLens, VerifyContext, type VisualDiffOptions, type VisualDiffResult, type ViteDeployRunnerInput, type WorkspaceAssertion, type WorkspaceAssertionResult, type WorkspaceInspector, type WorkspaceSnapshot, type WranglerDeployRunnerInput, adversarialJudge, aggregateJudgeVerdicts, aggregatePrReviewScore, analyzeAntiSlop, analyzeSeries, appendScorecard, assertCrossFamily, assertSingleBackend, attributeCounterfactuals, bisect, buildAgreementJudge, buildDriverSystemPrompt, buildReflectionPrompt, buildReviewerPrompt, canaryLeakView, canonicalize, causalAttribution, checkBehavioralCanary, checkCanaries, checkSlos, codeExecutionJudge, coherenceJudge, collectionPreserved, commentsForSource, commitBisect, compareReferenceReplay, compilerJudge, createAntiSlopJudge, createCustomJudge, createDefaultReviewer, createDomainExpertJudge, createIntentMatchJudge, createSandboxPool, crossTraceDiff, decideNextUserTurn, decideReferenceReplayPromotion, decideReferenceReplayRunPromotion, defaultJudges, defaultParseStudentLabel, defaultReferenceReplayMatcher, defaultRenderStudentPrompt, deployGateLayer, diffScorecard, discoverPersonas, distillPlaybook, estimateCost, estimateTokens, evaluateContract, evaluateHypothesis, evaluateOracles, executeScenario, expectAgent, exportRewardModel, extractAssetUrls, extractErrorCount, fieldAgreement, fileContains, fileExists, findAutoMatchNoExpectation, findConstructorCwdDropped, findFallbackToPass, findLiteralTruePass, findSkipCountsAsPass, flowLayer, formatBenchmarkReport, formatDriverReport, formatFindings, formatScorecardDiff, ghCliClient, precision as goldenPrecision, hashContent, hashJson, htmlContainsElement, httpGithubClient, inMemoryReferenceReplayStore, isModelPriced, isOtelConfigured, jsonShape, jsonlReferenceReplayStore, judgeFamily, keyPreserved, linterJudge, loadGoldScenarios, loadScorecard, loadScorerFromGrader, localCommandRunner, lowercaseMutator, matchGoldens, mergeLayerResults, multiToolchainLayer, notBlocked, paraphraseRobustness, paraphraseRobustnessScenarios, parseGoldJsonl, parseReflectionResponse, passOrthogonality, pixelDeltaRatio, politenessPrefixMutator, printDriverSummary, index as profile, promptBisect, proposeAutomatedPullRequest, proposeSynthesisTargets, recordRuns, recordRunsToScorecard, referenceReplayRunsToSteeringRows, referenceReplayScenarioToRunScore, regexMatches, renderMarkdownReport, renderPlaybookMarkdown, replayScorerOverCorpus, replayTraceThroughJudge, resetLockedAppendersForTesting, resolveModelPricing, rowCount, rowWhere, runAssertions, runBehavioralCanaries, runCanaries, runCounterfactual, runDistillation, runE2EWorkflow, runExpectations, runIntentMatchJudge, runJudgeFleet, runKeywordCoverageJudge, runKeywordCoverageJudgeUrl, runLiveProof, runReferenceReplay, runSelfPlay, scanForMuffledGates, scoreContinuity, scorePrReviewComments, scorePrReviewSource, scoreReferenceReplay, securityJudge, sentenceReorderMutator, signManifest, splitGold, statusAdvanced, summarizePrReviewBenchmark, testJudge, textInSnapshot, toLangfuseEnvelope, toPrometheusText, traceJudge, traceJudgeEnsemble, tracedAnalyzeTraces, typoMutator, urlContains, verifyManifest, visualDiff, viteDeployRunner, weightedRecall, whitespaceCollapseMutator, withJudgeRetry, withOtelPipeline, wranglerDeployRunner };
|