@tangle-network/agent-eval 0.103.1 → 0.104.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/dist/adapters/http.d.ts +3 -3
- package/dist/adapters/langchain.d.ts +3 -3
- package/dist/adapters/otel.d.ts +5 -5
- package/dist/analyst/index.d.ts +10 -10
- package/dist/analyst/index.js +8 -8
- package/dist/{analyze-runs-rF2eGxZ_.d.ts → analyze-runs-DJYpep3L.d.ts} +4 -4
- package/dist/belief-state/index.d.ts +4 -4
- package/dist/belief-state/index.js +3 -3
- package/dist/benchmarks/index.d.ts +3 -3
- package/dist/builder-eval/index.js +3 -3
- package/dist/campaign/index.d.ts +37 -17
- package/dist/campaign/index.js +14 -14
- package/dist/campaign/index.js.map +1 -1
- package/dist/{chunk-FRI6RG3P.js → chunk-2OGPXHOB.js} +3 -3
- package/dist/{chunk-VNM52AGA.js → chunk-3PFZBGMR.js} +3 -3
- package/dist/chunk-3PFZBGMR.js.map +1 -0
- package/dist/{chunk-SMCACT4Z.js → chunk-5PK3626Q.js} +4 -4
- package/dist/{chunk-UMEAR2FI.js → chunk-6HFYGEZ3.js} +3 -3
- package/dist/{chunk-CLS3374R.js → chunk-6MFFKPZ4.js} +2 -2
- package/dist/{chunk-REVYNR6C.js → chunk-6SK5VFYK.js} +2 -2
- package/dist/{chunk-YFYIOSNV.js → chunk-BBQILVBH.js} +5 -5
- package/dist/{chunk-BXZVBX3D.js → chunk-DBDRR6GF.js} +2 -2
- package/dist/{chunk-D5KBVULY.js → chunk-DRPIZQIT.js} +2 -2
- package/dist/{chunk-4DIJWVUT.js → chunk-DTJ6QUQB.js} +2 -2
- package/dist/{chunk-RQNOLV3I.js → chunk-FOUG2VVS.js} +2 -2
- package/dist/{chunk-CWNP4DV4.js → chunk-FUCQVFMU.js} +2 -2
- package/dist/{chunk-TCPP4Z5Q.js → chunk-GDZAWO2I.js} +4 -4
- package/dist/{chunk-ZOPLCJSY.js → chunk-HZHNRYHK.js} +2 -2
- package/dist/{chunk-BHCFJGL4.js → chunk-K7QEIHHJ.js} +2 -2
- package/dist/{chunk-6YVAN6R3.js → chunk-LIEJUH2I.js} +6 -6
- package/dist/{chunk-YBIGNSCZ.js → chunk-LNQEP766.js} +2 -2
- package/dist/{chunk-LMSQ6EFA.js → chunk-LOJ2QVCE.js} +4 -4
- package/dist/{chunk-LOZOZYHU.js → chunk-N22ZO7FV.js} +2 -2
- package/dist/{chunk-3BFEG2F6.js → chunk-ONWEPEDO.js} +1 -1
- package/dist/chunk-ONWEPEDO.js.map +1 -0
- package/dist/{chunk-L5TVEZFT.js → chunk-QRVS7MX4.js} +3 -3
- package/dist/{chunk-2MLIEQSN.js → chunk-RPDDVKI7.js} +2 -2
- package/dist/{chunk-SBCB6VZY.js → chunk-TT4KNT67.js} +2 -2
- package/dist/{chunk-3722PKPB.js → chunk-V7HNA47Z.js} +5 -5
- package/dist/{chunk-J5MUFXEY.js → chunk-VK6HBGAE.js} +3 -3
- package/dist/{chunk-QCBB6ZIU.js → chunk-X6NIXVOD.js} +2 -2
- package/dist/{chunk-AGYDMORK.js → chunk-XJYR7XFV.js} +2 -2
- package/dist/{chunk-HYGRFL7C.js → chunk-XMSYF4A7.js} +2 -2
- package/dist/{chunk-66T25VWE.js → chunk-Z7FK73R5.js} +5 -5
- package/dist/{chunk-2B4HPFXN.js → chunk-ZKIL3VOD.js} +5 -5
- package/dist/cli.js +3 -3
- package/dist/{code-agent-session-Cen-qD2y.d.ts → code-agent-session-rnJKlqmT.d.ts} +1 -1
- package/dist/contract/index.d.ts +20 -20
- package/dist/contract/index.js +14 -14
- package/dist/{control-KofK3gfG.d.ts → control-B8UthSBL.d.ts} +2 -2
- package/dist/control.d.ts +5 -5
- package/dist/control.js +4 -4
- package/dist/{corpus-CysLCiK4.d.ts → corpus-eBVwhCp1.d.ts} +1 -1
- package/dist/{dataset-BbGkaN2I.d.ts → dataset-DS7ytHZU.d.ts} +1 -1
- package/dist/{default-registry-Ez4cxuZ4.d.ts → default-registry-BswHCXnU.d.ts} +2 -2
- package/dist/diagnose.d.ts +5 -5
- package/dist/diagnose.js +5 -5
- package/dist/{errors-CzMUYo7b.d.ts → errors-oeQrLqXC.d.ts} +4 -0
- package/dist/{feedback-trajectory-BxY0cKfs.d.ts → feedback-trajectory-C9KCo8ag.d.ts} +1 -1
- package/dist/fuzz.js +1 -1
- package/dist/{gepa-COlCAkHN.d.ts → gepa-Dn0gtj9n.d.ts} +6 -2
- package/dist/governance/index.d.ts +3 -3
- package/dist/hosted/index.d.ts +5 -5
- package/dist/{index-unCSYRJJ.d.ts → index-C2CC_dry.d.ts} +1 -1
- package/dist/index.d.ts +32 -32
- package/dist/index.js +25 -25
- package/dist/{insight-report-DumCfEur.d.ts → insight-report-B4xrdwEK.d.ts} +1 -1
- package/dist/{integrity-D2t12mMw.d.ts → integrity-DqGZg3st.d.ts} +1 -1
- package/dist/{kind-factory-BLHxwX71.d.ts → kind-factory-DvIGo_cP.d.ts} +1 -1
- package/dist/{llm-client-Bj7g0rqu.d.ts → llm-client-DyqEH4jH.d.ts} +1 -1
- package/dist/meta-eval/index.d.ts +3 -3
- package/dist/meta-eval/index.js +3 -3
- package/dist/multishot/index.d.ts +59 -4
- package/dist/multishot/index.js +15 -10
- package/dist/multishot/index.js.map +1 -1
- package/dist/openapi.json +1 -1
- package/dist/pipelines/index.js +3 -3
- package/dist/{policy-edit-nUhuFtLF.d.ts → policy-edit-DQUXYMDm.d.ts} +3 -3
- package/dist/{pre-registration-CUOSGAZK.d.ts → pre-registration-D8Zr8-rr.d.ts} +4 -4
- package/dist/product-benchmark/index.d.ts +2 -2
- package/dist/product-benchmark/index.js +2 -2
- package/dist/{provenance-B2VsP0jP.d.ts → provenance-phkRBl4h.d.ts} +4 -4
- package/dist/{red-team-BWdoyleI.d.ts → red-team-KmmiqBlY.d.ts} +1 -1
- package/dist/{release-report-DkaCZ9k4.d.ts → release-report-DeJpsBiA.d.ts} +3 -3
- package/dist/reporting.d.ts +6 -6
- package/dist/reporting.js +5 -5
- package/dist/{researcher-SDAezfML.d.ts → researcher-Wc7dx6GM.d.ts} +4 -4
- package/dist/rl.d.ts +11 -11
- package/dist/rl.js +11 -11
- package/dist/{rubric-predictive-validity-ZGIlJNce.d.ts → rubric-predictive-validity-DPnyG-CE.d.ts} +1 -1
- package/dist/run-campaign-PNHAJIT2.js +12 -0
- package/dist/{run-record-CPfd1ARZ.d.ts → run-record-I-Z3JNvO.d.ts} +1 -1
- package/dist/{runtime-trajectory-DZ8ei-Jo.d.ts → runtime-trajectory-iW9IhV3e.d.ts} +1 -1
- package/dist/{semantic-concept-judge-C6mDEBIo.d.ts → semantic-concept-judge-BmNZPB_j.d.ts} +3 -3
- package/dist/{summary-report-Fc_YFJat.d.ts → summary-report-QMZVe3P-.d.ts} +1 -1
- package/dist/traces.d.ts +3 -3
- package/dist/traces.js +7 -7
- package/dist/{types-DA9yj-Jd.d.ts → types-D1ytG0Yg.d.ts} +2 -2
- package/dist/{types-Bihq6-a3.d.ts → types-Mh6I44ub.d.ts} +1 -1
- package/dist/wire/index.d.ts +3 -3
- package/dist/wire/index.js +3 -3
- package/dist/workflow/index.d.ts +8 -8
- package/dist/workflow/index.js +4 -4
- package/package.json +1 -1
- package/dist/chunk-3BFEG2F6.js.map +0 -1
- package/dist/chunk-VNM52AGA.js.map +0 -1
- package/dist/run-campaign-DAHKO5CT.js +0 -12
- /package/dist/{chunk-FRI6RG3P.js.map → chunk-2OGPXHOB.js.map} +0 -0
- /package/dist/{chunk-SMCACT4Z.js.map → chunk-5PK3626Q.js.map} +0 -0
- /package/dist/{chunk-UMEAR2FI.js.map → chunk-6HFYGEZ3.js.map} +0 -0
- /package/dist/{chunk-CLS3374R.js.map → chunk-6MFFKPZ4.js.map} +0 -0
- /package/dist/{chunk-REVYNR6C.js.map → chunk-6SK5VFYK.js.map} +0 -0
- /package/dist/{chunk-YFYIOSNV.js.map → chunk-BBQILVBH.js.map} +0 -0
- /package/dist/{chunk-BXZVBX3D.js.map → chunk-DBDRR6GF.js.map} +0 -0
- /package/dist/{chunk-D5KBVULY.js.map → chunk-DRPIZQIT.js.map} +0 -0
- /package/dist/{chunk-4DIJWVUT.js.map → chunk-DTJ6QUQB.js.map} +0 -0
- /package/dist/{chunk-RQNOLV3I.js.map → chunk-FOUG2VVS.js.map} +0 -0
- /package/dist/{chunk-CWNP4DV4.js.map → chunk-FUCQVFMU.js.map} +0 -0
- /package/dist/{chunk-TCPP4Z5Q.js.map → chunk-GDZAWO2I.js.map} +0 -0
- /package/dist/{chunk-ZOPLCJSY.js.map → chunk-HZHNRYHK.js.map} +0 -0
- /package/dist/{chunk-BHCFJGL4.js.map → chunk-K7QEIHHJ.js.map} +0 -0
- /package/dist/{chunk-6YVAN6R3.js.map → chunk-LIEJUH2I.js.map} +0 -0
- /package/dist/{chunk-YBIGNSCZ.js.map → chunk-LNQEP766.js.map} +0 -0
- /package/dist/{chunk-LMSQ6EFA.js.map → chunk-LOJ2QVCE.js.map} +0 -0
- /package/dist/{chunk-LOZOZYHU.js.map → chunk-N22ZO7FV.js.map} +0 -0
- /package/dist/{chunk-L5TVEZFT.js.map → chunk-QRVS7MX4.js.map} +0 -0
- /package/dist/{chunk-2MLIEQSN.js.map → chunk-RPDDVKI7.js.map} +0 -0
- /package/dist/{chunk-SBCB6VZY.js.map → chunk-TT4KNT67.js.map} +0 -0
- /package/dist/{chunk-3722PKPB.js.map → chunk-V7HNA47Z.js.map} +0 -0
- /package/dist/{chunk-J5MUFXEY.js.map → chunk-VK6HBGAE.js.map} +0 -0
- /package/dist/{chunk-QCBB6ZIU.js.map → chunk-X6NIXVOD.js.map} +0 -0
- /package/dist/{chunk-AGYDMORK.js.map → chunk-XJYR7XFV.js.map} +0 -0
- /package/dist/{chunk-HYGRFL7C.js.map → chunk-XMSYF4A7.js.map} +0 -0
- /package/dist/{chunk-66T25VWE.js.map → chunk-Z7FK73R5.js.map} +0 -0
- /package/dist/{chunk-2B4HPFXN.js.map → chunk-ZKIL3VOD.js.map} +0 -0
- /package/dist/{run-campaign-DAHKO5CT.js.map → run-campaign-PNHAJIT2.js.map} +0 -0
package/dist/index.d.ts
CHANGED
|
@@ -1,28 +1,28 @@
|
|
|
1
|
-
export { A as ActionExecutionPolicy, a as ActionPolicyDecision, C as ControlRunToRunRecordOptions, L as LlmJsonCall, b as LlmReviewerConfig, P as ProposeFn, c as ProposeInput, d as ProposeOutput, e as ProposeReviewConfig, f as ProposeReviewControlAction, g as ProposeReviewControlConfig, h as ProposeReviewControlResult, i as ProposeReviewControlState, j as ProposeReviewReport, k as ProposeReviewShot, R as Review, l as ReviewFn, m as ReviewInput, n as ReviewMemoryEntry, o as ReviewMemoryStore, p as RunEvidenceMetadata, V as Verification, q as VerifyFn, r as controlFailureClassFromVerification, s as controlRunToRunRecord, t as createLlmReviewer, u as evaluateActionPolicy, v as inMemoryReviewStore, w as jsonlReviewStore, x as runProposeReview, y as runProposeReviewAsControlLoop, z as scoreFromEvals } from './control-
|
|
2
|
-
import { R as RunRecord, b as RunSplitTag } from './run-record-
|
|
3
|
-
export { f as AGENT_PROFILE_KINDS, g as AgentInterfaceProfileLike, A as AgentProfileCell, e as AgentProfileCellInput, h as AgentProfileCellSchemaVersion, i as AgentProfileCellValidationError, j as AgentProfileDimensionValue, k as AgentProfileHarness, a as AgentProfileJson, l as AgentProfileKind, m as AgentProfileSource, n as AgentProfileSourceInput, J as JudgeScoresRecord, d as RunJudgeMetadata, o as RunOutcome, p as RunRecordValidationError, c as RunTokenUsage, q as agentProfileCellHashMaterial, r as agentProfileCellKey, s as assertRunAgentProfileCell, t as buildAgentInterfaceProfileCell, u as buildAgentProfileCell, v as groupRunsByAgentProfileCell, w as isRunRecord, x as modelHasSnapshot, y as parseRunRecordSafe, z as requireAgentProfileCell, B as roundTripRunRecord, C as toAgentProfileJson, D as validateAgentProfileCell, E as validateRunRecord, F as verifyAgentProfileCell } from './run-record-
|
|
4
|
-
import { B as BehavioralMetrics } from './semantic-concept-judge-
|
|
5
|
-
export { x as ConceptComplexity, y as ConceptFinding, z as ConceptSpec, A as ConceptWeightStrategy, C as CreateAnalystAiConfig, E as DEFAULT_COMPLEXITY_WEIGHTS, D as DEFAULT_TRACE_ANALYST_KINDS, b as DiffPolicy, F as FAILURE_MODE_KIND_SPEC, e as FindingSubject, f as FindingSubjectKind, h as FindingsDiff, i as FindingsStore, I as IMPROVEMENT_KIND_SPEC, j as KNOWLEDGE_GAP_KIND_SPEC, k as KNOWLEDGE_POISONING_KIND_SPEC, P as PersistedFinding, G as SEMANTIC_CONCEPT_JUDGE_VERSION, l as SKILL_USAGE_ANALYST, a as SemanticConceptJudgeInput, S as SemanticConceptJudgeOptions, H as SemanticConceptJudgeResult, m as SkillUsageAnalyst, J as SuboptimalCode, L as SuboptimalSignal, M as computeTraceMetrics, r as createAnalystAi, N as createSemanticConceptJudge, s as defaultIsMaterial, t as diffFindings, O as runSemanticConceptJudge } from './semantic-concept-judge-
|
|
6
|
-
import { l as ChatRequest, p as CreateChatClientOpts } from './types-
|
|
7
|
-
export { a as Analyst, b as AnalystContext, g as AnalystCost, A as AnalystFinding, i as AnalystInputKind, j as AnalystRequirements, f as AnalystRunEvent, e as AnalystRunInputs, d as AnalystRunResult, c as AnalystRunSummary, h as AnalystSeverity, k as ChatCallOpts, C as ChatClient, m as ChatResponse, n as ChatTransport, o as CliBridgeTransportOpts, D as DirectProviderTransportOpts, E as EvidenceRef, M as MockTransportOpts, R as RouterTransportOpts, S as SandboxSdkTransportOpts, q as computeFindingId, r as createChatClient, s as makeFinding } from './types-
|
|
8
|
-
export { a as AnalystHooks, A as AnalystRegistry, b as AnalystRegistryOptions, B as BudgetPolicy, D as DefaultAnalystRegistryOptions, R as RegistryRunOpts, c as buildDefaultAnalystRegistry } from './default-registry-
|
|
9
|
-
export { C as CreateTraceAnalystKindOpts, a as RawAnalystFinding, c as TraceAnalystGolden, T as TraceAnalystKindSpec, d as createTraceAnalystKind, r as renderPriorFindings } from './kind-factory-
|
|
10
|
-
export { F as FindingToPolicyEditOptions, P as POLICY_EDIT_AXES, a as POLICY_EDIT_TARGET_SURFACES, b as PolicyEdit, c as PolicyEditAdmission, d as PolicyEditAdmissionOptions, e as PolicyEditAxis, f as PolicyEditChange, g as PolicyEditExpectedGain, h as PolicyEditGainDirection, i as PolicyEditGainUnit, j as PolicyEditInit, k as PolicyEditRisk, l as PolicyEditSchemaVersion, m as PolicyEditSource, n as PolicyEditTarget, o as PolicyEditTargetSurface, p as PolicyEditValidationError, q as admitPolicyEdit, r as applyPolicyEditToSurface, s as computePolicyEditId, t as isPolicyEdit, u as makePolicyEdit, v as policyEditFromFinding, w as policyEditsFromFindings, x as scorePolicyEditReadiness, y as validatePolicyEdit } from './policy-edit-
|
|
1
|
+
export { A as ActionExecutionPolicy, a as ActionPolicyDecision, C as ControlRunToRunRecordOptions, L as LlmJsonCall, b as LlmReviewerConfig, P as ProposeFn, c as ProposeInput, d as ProposeOutput, e as ProposeReviewConfig, f as ProposeReviewControlAction, g as ProposeReviewControlConfig, h as ProposeReviewControlResult, i as ProposeReviewControlState, j as ProposeReviewReport, k as ProposeReviewShot, R as Review, l as ReviewFn, m as ReviewInput, n as ReviewMemoryEntry, o as ReviewMemoryStore, p as RunEvidenceMetadata, V as Verification, q as VerifyFn, r as controlFailureClassFromVerification, s as controlRunToRunRecord, t as createLlmReviewer, u as evaluateActionPolicy, v as inMemoryReviewStore, w as jsonlReviewStore, x as runProposeReview, y as runProposeReviewAsControlLoop, z as scoreFromEvals } from './control-B8UthSBL.js';
|
|
2
|
+
import { R as RunRecord, b as RunSplitTag } from './run-record-I-Z3JNvO.js';
|
|
3
|
+
export { f as AGENT_PROFILE_KINDS, g as AgentInterfaceProfileLike, A as AgentProfileCell, e as AgentProfileCellInput, h as AgentProfileCellSchemaVersion, i as AgentProfileCellValidationError, j as AgentProfileDimensionValue, k as AgentProfileHarness, a as AgentProfileJson, l as AgentProfileKind, m as AgentProfileSource, n as AgentProfileSourceInput, J as JudgeScoresRecord, d as RunJudgeMetadata, o as RunOutcome, p as RunRecordValidationError, c as RunTokenUsage, q as agentProfileCellHashMaterial, r as agentProfileCellKey, s as assertRunAgentProfileCell, t as buildAgentInterfaceProfileCell, u as buildAgentProfileCell, v as groupRunsByAgentProfileCell, w as isRunRecord, x as modelHasSnapshot, y as parseRunRecordSafe, z as requireAgentProfileCell, B as roundTripRunRecord, C as toAgentProfileJson, D as validateAgentProfileCell, E as validateRunRecord, F as verifyAgentProfileCell } from './run-record-I-Z3JNvO.js';
|
|
4
|
+
import { B as BehavioralMetrics } from './semantic-concept-judge-BmNZPB_j.js';
|
|
5
|
+
export { x as ConceptComplexity, y as ConceptFinding, z as ConceptSpec, A as ConceptWeightStrategy, C as CreateAnalystAiConfig, E as DEFAULT_COMPLEXITY_WEIGHTS, D as DEFAULT_TRACE_ANALYST_KINDS, b as DiffPolicy, F as FAILURE_MODE_KIND_SPEC, e as FindingSubject, f as FindingSubjectKind, h as FindingsDiff, i as FindingsStore, I as IMPROVEMENT_KIND_SPEC, j as KNOWLEDGE_GAP_KIND_SPEC, k as KNOWLEDGE_POISONING_KIND_SPEC, P as PersistedFinding, G as SEMANTIC_CONCEPT_JUDGE_VERSION, l as SKILL_USAGE_ANALYST, a as SemanticConceptJudgeInput, S as SemanticConceptJudgeOptions, H as SemanticConceptJudgeResult, m as SkillUsageAnalyst, J as SuboptimalCode, L as SuboptimalSignal, M as computeTraceMetrics, r as createAnalystAi, N as createSemanticConceptJudge, s as defaultIsMaterial, t as diffFindings, O as runSemanticConceptJudge } from './semantic-concept-judge-BmNZPB_j.js';
|
|
6
|
+
import { l as ChatRequest, p as CreateChatClientOpts } from './types-D1ytG0Yg.js';
|
|
7
|
+
export { a as Analyst, b as AnalystContext, g as AnalystCost, A as AnalystFinding, i as AnalystInputKind, j as AnalystRequirements, f as AnalystRunEvent, e as AnalystRunInputs, d as AnalystRunResult, c as AnalystRunSummary, h as AnalystSeverity, k as ChatCallOpts, C as ChatClient, m as ChatResponse, n as ChatTransport, o as CliBridgeTransportOpts, D as DirectProviderTransportOpts, E as EvidenceRef, M as MockTransportOpts, R as RouterTransportOpts, S as SandboxSdkTransportOpts, q as computeFindingId, r as createChatClient, s as makeFinding } from './types-D1ytG0Yg.js';
|
|
8
|
+
export { a as AnalystHooks, A as AnalystRegistry, b as AnalystRegistryOptions, B as BudgetPolicy, D as DefaultAnalystRegistryOptions, R as RegistryRunOpts, c as buildDefaultAnalystRegistry } from './default-registry-BswHCXnU.js';
|
|
9
|
+
export { C as CreateTraceAnalystKindOpts, a as RawAnalystFinding, c as TraceAnalystGolden, T as TraceAnalystKindSpec, d as createTraceAnalystKind, r as renderPriorFindings } from './kind-factory-DvIGo_cP.js';
|
|
10
|
+
export { F as FindingToPolicyEditOptions, P as POLICY_EDIT_AXES, a as POLICY_EDIT_TARGET_SURFACES, b as PolicyEdit, c as PolicyEditAdmission, d as PolicyEditAdmissionOptions, e as PolicyEditAxis, f as PolicyEditChange, g as PolicyEditExpectedGain, h as PolicyEditGainDirection, i as PolicyEditGainUnit, j as PolicyEditInit, k as PolicyEditRisk, l as PolicyEditSchemaVersion, m as PolicyEditSource, n as PolicyEditTarget, o as PolicyEditTargetSurface, p as PolicyEditValidationError, q as admitPolicyEdit, r as applyPolicyEditToSurface, s as computePolicyEditId, t as isPolicyEdit, u as makePolicyEdit, v as policyEditFromFinding, w as policyEditsFromFindings, x as scorePolicyEditReadiness, y as validatePolicyEdit } from './policy-edit-DQUXYMDm.js';
|
|
11
11
|
import { TCloud } from '@tangle-network/tcloud';
|
|
12
12
|
import { B as BenchmarkRunnerConfig, S as Scenario, c as BenchmarkReport, P as ProductClientConfig, C as CheckResult, T as TestResult, d as PersonaConfig, D as DriverResult, e as DriverState, b as JudgeFn, f as CollectedArtifacts, g as ScenarioResult, h as TurnMetrics, i as ScenarioFile, j as CompletionCriterion } from './types-C7DGg5ex.js';
|
|
13
13
|
export { A as ArtifactCheck, k as ArtifactResult, E as EvalResult, F as FeedbackPattern, l as JudgeConfig, a as JudgeInput, m as JudgeRubric, J as JudgeScore, n as PersonaRigor, R as RouteMap, o as RubricDimension, p as Turn, q as TurnResult } from './types-C7DGg5ex.js';
|
|
14
14
|
export { c as ControlActionFailureMode, d as ControlActionOutcome, e as ControlBudget, f as ControlContext, g as ControlDecision, C as ControlEvalResult, a as ControlRunResult, h as ControlRuntimeConfig, i as ControlRuntimeError, j as ControlSeverity, b as ControlStep, k as ControlStopPolicies, S as StopDecision, l as allCriticalPassed, o as objectiveEval, r as runAgentControlLoop, s as stopOnNoProgress, m as stopOnRepeatedAction, n as subjectiveEval } from './control-runtime-Acf9CGhw.js';
|
|
15
15
|
import { F as FailureClass, T as ToolSpan, h as BudgetSpec, B as BudgetLedgerEntry, R as Run, L as LlmSpan } from './schema-m0gsnbt3.js';
|
|
16
16
|
export { A as Artifact, E as EventKind, i as FAILURE_CLASSES, G as GenericSpan, J as JudgeSpan, M as Message, d as RetrievalSpan, g as RunLayer, f as RunStatus, e as SandboxSpan, S as Span, j as SpanBase, c as SpanKind, k as SpanStatus, l as TRACE_SCHEMA_VERSION, a as TraceEvent, m as isJudgeSpan, n as isLlmSpan, o as isRetrievalSpan, p as isSandboxSpan, q as isToolSpan } from './schema-m0gsnbt3.js';
|
|
17
|
-
import { A as AgentEvalError, J as JudgeError, a as ConfigError } from './errors-
|
|
18
|
-
export { b as AgentEvalErrorCode, C as CaptureIntegrityError, N as NotFoundError, R as ReplayError, V as ValidationError, c as VerificationError } from './errors-
|
|
19
|
-
import { b as FeedbackLabel, F as FeedbackTrajectoryStore, a as FeedbackTrajectory } from './feedback-trajectory-
|
|
20
|
-
export { c as FeedbackArtifactType, d as FeedbackAttempt, e as FeedbackLabelKind, f as FeedbackLabelSource, g as FeedbackOptimizerRow, h as FeedbackOutcome, i as FeedbackReplayAdapter, j as FeedbackReplayResult, k as FeedbackSeverity, l as FeedbackSplitPolicy, m as FeedbackTask, n as FeedbackTrajectoryFilter, o as FileSystemFeedbackTrajectoryStore, I as InMemoryFeedbackTrajectoryStore, P as PreferenceMemoryEntry, p as ProposedSideEffect, q as assignFeedbackSplit, r as controlRunToFeedbackTrajectory, s as createFeedbackTrajectory, t as feedbackTrajectoriesToDatasetScenarios, u as feedbackTrajectoriesToOptimizerRows, v as feedbackTrajectoryToDatasetScenario, w as feedbackTrajectoryToOptimizerRow, x as parseFeedbackTrajectoriesJsonl, y as renderPreferenceMemoryMarkdown, z as replayFeedbackTrajectories, A as replayFeedbackTrajectory, B as serializeFeedbackTrajectoriesJsonl, C as summarizePreferenceMemory, D as withAssignedFeedbackSplit } from './feedback-trajectory-
|
|
21
|
-
import { b as CorrectnessChecker } from './pre-registration-
|
|
22
|
-
export { A as ArtifactCheckArtifact, d as ArtifactEventLike, e as ArtifactValidator, f as BackendIntegrityError, B as BackendIntegrityReport, C as CompletionRequirement, a as CompletionVerdict, H as HypothesisManifest, g as HypothesisResult, h as LlmCorrectnessCheckerOpts, L as LlmJudgeDimension, c as LlmJudgeOptions, i as ProducedProposal, P as ProducedState, j as ProposalEventLike, k as RequirementCheck, R as RuntimeEventLike, m as SatisfiedBy, S as SignedManifest, n as SignedManifestAlgo, T as TaskGold, o as ToolCallEventLike, V as ValidationContext, p as ValidationIssue, q as ValidationResult, r as assertRealBackend, s as byteLengthRange, t as canonicalize, u as completionVerdict, v as composeValidators, w as containsAll, x as createLlmCorrectnessChecker, y as createTokenRecallChecker, z as evaluateHypothesis, D as extractProducedState, E as hashJson, F as jsonHasKeys, l as llmJudge, G as parseCorrectnessResponse, I as regexMatch, J as signManifest, K as summarizeBackendIntegrity, M as verifyCompletion, N as verifyManifest } from './pre-registration-
|
|
17
|
+
import { A as AgentEvalError, J as JudgeError, a as ConfigError } from './errors-oeQrLqXC.js';
|
|
18
|
+
export { b as AgentEvalErrorCode, C as CaptureIntegrityError, N as NotFoundError, R as ReplayError, V as ValidationError, c as VerificationError } from './errors-oeQrLqXC.js';
|
|
19
|
+
import { b as FeedbackLabel, F as FeedbackTrajectoryStore, a as FeedbackTrajectory } from './feedback-trajectory-C9KCo8ag.js';
|
|
20
|
+
export { c as FeedbackArtifactType, d as FeedbackAttempt, e as FeedbackLabelKind, f as FeedbackLabelSource, g as FeedbackOptimizerRow, h as FeedbackOutcome, i as FeedbackReplayAdapter, j as FeedbackReplayResult, k as FeedbackSeverity, l as FeedbackSplitPolicy, m as FeedbackTask, n as FeedbackTrajectoryFilter, o as FileSystemFeedbackTrajectoryStore, I as InMemoryFeedbackTrajectoryStore, P as PreferenceMemoryEntry, p as ProposedSideEffect, q as assignFeedbackSplit, r as controlRunToFeedbackTrajectory, s as createFeedbackTrajectory, t as feedbackTrajectoriesToDatasetScenarios, u as feedbackTrajectoriesToOptimizerRows, v as feedbackTrajectoryToDatasetScenario, w as feedbackTrajectoryToOptimizerRow, x as parseFeedbackTrajectoriesJsonl, y as renderPreferenceMemoryMarkdown, z as replayFeedbackTrajectories, A as replayFeedbackTrajectory, B as serializeFeedbackTrajectoriesJsonl, C as summarizePreferenceMemory, D as withAssignedFeedbackSplit } from './feedback-trajectory-C9KCo8ag.js';
|
|
21
|
+
import { b as CorrectnessChecker } from './pre-registration-D8Zr8-rr.js';
|
|
22
|
+
export { A as ArtifactCheckArtifact, d as ArtifactEventLike, e as ArtifactValidator, f as BackendIntegrityError, B as BackendIntegrityReport, C as CompletionRequirement, a as CompletionVerdict, H as HypothesisManifest, g as HypothesisResult, h as LlmCorrectnessCheckerOpts, L as LlmJudgeDimension, c as LlmJudgeOptions, i as ProducedProposal, P as ProducedState, j as ProposalEventLike, k as RequirementCheck, R as RuntimeEventLike, m as SatisfiedBy, S as SignedManifest, n as SignedManifestAlgo, T as TaskGold, o as ToolCallEventLike, V as ValidationContext, p as ValidationIssue, q as ValidationResult, r as assertRealBackend, s as byteLengthRange, t as canonicalize, u as completionVerdict, v as composeValidators, w as containsAll, x as createLlmCorrectnessChecker, y as createTokenRecallChecker, z as evaluateHypothesis, D as extractProducedState, E as hashJson, F as jsonHasKeys, l as llmJudge, G as parseCorrectnessResponse, I as regexMatch, J as signManifest, K as summarizeBackendIntegrity, M as verifyCompletion, N as verifyManifest } from './pre-registration-D8Zr8-rr.js';
|
|
23
23
|
export { DataAcquisitionPlan, KnowledgeAcquisitionMode, KnowledgeBundle, KnowledgeFallbackPolicy, KnowledgeFreshness, KnowledgeImportance, KnowledgeReadinessReport, KnowledgeRecommendedAction, KnowledgeRequirement, KnowledgeRequirementCategory, KnowledgeResponsibleSurface, KnowledgeSensitivity, ScoreKnowledgeReadinessOptions, UserQuestion, acquisitionPlansForKnowledgeGaps, blockingKnowledgeEval, knowledgeReadinessTracePayload, scoreKnowledgeReadiness, userQuestionsForKnowledgeGaps } from './knowledge/index.js';
|
|
24
|
-
import { h as ReleaseConfidenceThresholds, f as ReleaseConfidenceScorecard } from './release-report-
|
|
25
|
-
export { A as ActionableSideInfo, o as AsiSeverity, B as BootstrapOptions, a as BootstrapResult, J as JudgeReplayGateArgs, R as ReleaseConfidenceAxis, b as ReleaseConfidenceAxisName, c as ReleaseConfidenceInput, d as ReleaseConfidenceIssue, e as ReleaseConfidenceMetrics, g as ReleaseConfidenceStatus, i as ReleaseTraceEvidence, j as RenderReleaseReportOptions, V as Verdict, k as assertReleaseConfidence, l as bootstrapCi, m as evaluateReleaseConfidence, n as judgeReplayGate, r as renderReleaseReport } from './release-report-
|
|
24
|
+
import { h as ReleaseConfidenceThresholds, f as ReleaseConfidenceScorecard } from './release-report-DeJpsBiA.js';
|
|
25
|
+
export { A as ActionableSideInfo, o as AsiSeverity, B as BootstrapOptions, a as BootstrapResult, J as JudgeReplayGateArgs, R as ReleaseConfidenceAxis, b as ReleaseConfidenceAxisName, c as ReleaseConfidenceInput, d as ReleaseConfidenceIssue, e as ReleaseConfidenceMetrics, g as ReleaseConfidenceStatus, i as ReleaseTraceEvidence, j as RenderReleaseReportOptions, V as Verdict, k as assertReleaseConfidence, l as bootstrapCi, m as evaluateReleaseConfidence, n as judgeReplayGate, r as renderReleaseReport } from './release-report-DeJpsBiA.js';
|
|
26
26
|
export { c as CliffsMagnitude, d as CorpusAgreementOptions, e as CorpusAgreementPerDimension, C as CorpusAgreementReport, f as CorpusScoreRecord, g as EProcess, h as EProcessOptions, E as EProcessState, i as EProcessStep, M as McNemarResult, P as PairedBootstrapOptions, a as PairedBootstrapResult, j as ProportionInterval, R as RiskDifferenceResult, W as WeightedCompositeInput, k as WeightedCompositeResult, b as benjaminiHochberg, l as bonferroni, m as cliffsDelta, n as cohensD, o as confidenceInterval, q as corpusInterRaterAgreement, r as corpusInterRaterAgreementFromJudgeScores, s as eProcess, t as interRaterReliability, u as interpretCliffs, v as mannWhitneyU, x as mcnemar, y as mcnemarPower, z as mcnemarRequiredN, A as mulberry32, B as normalizeScores, p as pairedBootstrap, D as pairedMde, F as pairedRiskDifference, G as pairedTTest, H as partialCredit, I as passAtK, J as pearsonR, K as ranks, L as requiredSampleSize, N as spearmanR, O as weightedComposite, Q as weightedMean, w as wilcoxonSignedRank, S as wilson } from './statistics-D88peojY.js';
|
|
27
27
|
import { OtelExporter, OtelExportConfig } from './traces.js';
|
|
28
28
|
export { CaptureFetchContext, CaptureFetchOptions, ExportableSpan, ExtractedUsage, FlattenOtlpOptions, LLM_CACHED_TOKENS, LLM_CACHED_TOKEN_ATTR_KEYS, LLM_COST_ATTR_KEYS, LLM_COST_USD, LLM_INPUT_TOKENS, LLM_INPUT_TOKEN_ATTR_KEYS, LLM_MODEL_ATTR_KEYS, LLM_MODEL_NAME, LLM_OUTPUT_TOKENS, LLM_OUTPUT_TOKEN_ATTR_KEYS, OPENINFERENCE_SPAN_KIND, OTEL_AGENT_EVAL_SCOPE, OtlpExport, OtlpFileTraceStore, OtlpFileTraceStoreOptions, OtlpFlatLine, OtlpResourceSpans, OtlpSpan, OtlpToRunRecordsOptions, OtlpTraceRunRecord, ProjectedOtlpSpan, ReplayCache, ReplayCacheEntry, ReplayCacheMissError, ReplayCacheStats, ReplayFetchOptions, SPAN_KIND_ATTR_KEYS, SpanNotFoundError, TOOL_NAME, TOOL_NAME_ATTR_KEYS, TRACE_ANALYST_ACTOR_DESCRIPTION, TRACE_ANALYST_ACTOR_DESCRIPTION_VERSION, TRACE_ANALYST_SUBAGENT_DESCRIPTION, TraceAggregate, TraceAnalystHookOptions, TraceFileMissingError, TraceInsightContext, TraceInsightFinding, TraceInsightPanelRole, TraceInsightPromptInput, TraceInsightQualityGate, TraceInsightQuestion, TraceInsightReadiness, TraceInsightSuite, TraceInsightTask, TraceNotFoundError, TraceStoreSource, TraceStoreToOtlpOptions, TracesToOtlpResult, asNumber, asString, buildTraceAnalystTools, buildTraceInsightContext, buildTraceInsightPrompt, captureFetchToRawSink, convertTraceStoresToOtlp, createOtelExporter, createOtelTracingStore, createReplayFetch, defaultTraceInsightPanel, describeTraceInsightScope, domainEvidencePattern, exportRunAsOtlp, extractOtlpAttributes, extractUsage, extractUsageFromResponse, extractUsageFromSse, firstNumberAttr, firstStringAttr, flattenOtlpExportToNdjson, inferDomainKeywords, inferOtlpKind, iterateRawCalls, otelRunCompleteHook, otlpToRunRecords, otlpToTraceRunRecords, planTraceInsightQuestions, projectOtlpFlatLine, readOtlpStatus, scoreTraceInsightReadiness, stringField, tokenizeDomainWords, traceAnalystFunctionGroup, traceAnalystOnRunComplete, traceSpanKindToOpenInferenceKind } from './traces.js';
|
|
@@ -30,8 +30,8 @@ import { a as AnalyzeTracesInput, A as AnalyzeTracesOptions, b as AnalyzeTracesR
|
|
|
30
30
|
export { c as AnalyzeTracesTurnSnapshot, d as analyzeTraces } from './analyst-C8HHvfJp.js';
|
|
31
31
|
import { a as TraceAnalystSpan } from './store-C1YxJDEK.js';
|
|
32
32
|
export { D as DEFAULT_TRACE_ANALYST_BUDGETS, b as DatasetOverview, E as ErrorCluster, Q as QueryTracesPage, S as SearchSpanResult, c as SearchTraceResult, d as SpanMatchRecord, e as TRACE_ANALYST_TRUNCATION_MARKER_PREFIX, T as TraceAnalysisStore, f as TraceAnalystByteBudgets, g as TraceAnalystFilters, h as TraceAnalystSpanKind, i as TraceAnalystSpanStatus, j as TraceAnalystTraceSummary, V as ViewSpansResult, k as ViewTraceOversized, l as ViewTraceResult } from './store-C1YxJDEK.js';
|
|
33
|
-
import { a as JudgeConfig, J as JudgeScore, S as Scenario$1, g as Gate } from './types-
|
|
34
|
-
import { A as AnalyzeRunsOptions } from './analyze-runs-
|
|
33
|
+
import { a as JudgeConfig, J as JudgeScore, S as Scenario$1, g as Gate } from './types-Mh6I44ub.js';
|
|
34
|
+
import { A as AnalyzeRunsOptions } from './analyze-runs-DJYpep3L.js';
|
|
35
35
|
import { S as SteeringBundle } from './harness-optimizer-mOl9XX_O.js';
|
|
36
36
|
export { D as DEFAULT_HARNESS_OBJECTIVES, H as HarnessAdapter, a as HarnessExperimentConfig, b as HarnessExperimentResult, c as HarnessIntervention, d as HarnessRunRequest, e as HarnessRunResult, f as HarnessScenario, g as HarnessSelection, h as HarnessVariant, i as HarnessVariantReport, M as MeasurementPolicy, j as SteeringDelta, k as SteeringRolePrompt, W as WorkflowTopology, m as mergeSteeringBundle, r as renderSteeringText, l as runHarnessExperiment, s as selectHarnessVariant, n as summarizeHarnessResults } from './harness-optimizer-mOl9XX_O.js';
|
|
37
37
|
import { S as SandboxDriver, H as HarnessConfig, a as SandboxHarnessResult } from './test-graded-scenario-DeODGLra.js';
|
|
@@ -40,14 +40,14 @@ import { b as RunScoreWeights, R as RunScore } from './run-critic-CmMf05uV.js';
|
|
|
40
40
|
export { D as DEFAULT_RUN_SCORE_WEIGHTS, c as RunCritic, d as RunCriticOptions, a as RunTrace, e as aggregateRunScore, f as clamp01 } from './run-critic-CmMf05uV.js';
|
|
41
41
|
import { T as TraceEmitter } from './emitter-C2rqGH_l.js';
|
|
42
42
|
export { R as RunCompleteHook, a as RunCompleteHookContext, S as SpanHandle, b as TraceEmitterOptions, l as llmSpanFromProvider } from './emitter-C2rqGH_l.js';
|
|
43
|
-
export { b as RunIntegrityError, R as RunIntegrityExpectations, c as RunIntegrityIssue, d as RunIntegrityIssueCode, a as RunIntegrityReport, e as assertRunCaptured, t as throwIfRunIncomplete } from './integrity-
|
|
43
|
+
export { b as RunIntegrityError, R as RunIntegrityExpectations, c as RunIntegrityIssue, d as RunIntegrityIssueCode, a as RunIntegrityReport, e as assertRunCaptured, t as throwIfRunIncomplete } from './integrity-DqGZg3st.js';
|
|
44
44
|
export { a as aggregateLlm, b as argHash, g as groupBy, j as judgeSpans, l as llmSpans, r as runFailureClass, c as runsForScenario, t as toolSpans } from './query-0aTmbmQe.js';
|
|
45
45
|
export { F as FileSystemRawProviderSink, a as FileSystemRawProviderSinkOptions, I as InMemoryRawProviderSink, b as InMemoryRawProviderSinkOptions, N as NoopRawProviderSink, P as ProviderRedactor, c as RawProviderDirection, d as RawProviderEvent, R as RawProviderSink, e as RawProviderSinkFilter, f as defaultProviderRedactor, p as providerFromBaseUrl } from './raw-provider-sink-C46HDghv.js';
|
|
46
46
|
export { D as DEFAULT_REDACTION_RULES, b as REDACTION_VERSION, a as RedactionReport, R as RedactionRule, r as redactString, c as redactValue } from './redact-B40YG2M_.js';
|
|
47
47
|
import { T as TraceStore, R as RunFilter } from './store-BcFXE6LG.js';
|
|
48
48
|
export { E as EventFilter, F as FileSystemTraceStore, a as FileSystemTraceStoreOptions, I as InMemoryTraceStore, S as SpanFilter } from './store-BcFXE6LG.js';
|
|
49
49
|
export { D as DEFAULT_FAILURE_RULES, b as FailureClassification, c as FailureContext, d as FailureRule, e as classifyFailure } from './failure-cluster-DH9Flgcf.js';
|
|
50
|
-
export { P as ProjectRuntimeTrajectoryEvidenceOptions, a as RuntimeTrajectoryEvidenceProjection, b as RuntimeTrajectoryEvidenceSummary, c as RuntimeTrajectoryHookEvent, R as RuntimeTrajectoryRecord, d as RuntimeTrajectoryRunRecord, p as parseRuntimeTrajectoryHookEvent, e as projectRuntimeTrajectoryEvidence } from './runtime-trajectory-
|
|
50
|
+
export { P as ProjectRuntimeTrajectoryEvidenceOptions, a as RuntimeTrajectoryEvidenceProjection, b as RuntimeTrajectoryEvidenceSummary, c as RuntimeTrajectoryHookEvent, R as RuntimeTrajectoryRecord, d as RuntimeTrajectoryRunRecord, p as parseRuntimeTrajectoryHookEvent, e as projectRuntimeTrajectoryEvidence } from './runtime-trajectory-iW9IhV3e.js';
|
|
51
51
|
import { a as BaselineReport } from './baseline-Bbid3WoO.js';
|
|
52
52
|
export { B as BaselineOptions, M as MetricSamples, b as MetricVerdict, T as ToolStats, d as ToolUseMetrics, e as ToolUseOptions, f as compareToBaseline, c as computeToolUseMetrics, i as iqr, w as welchsTTest } from './baseline-Bbid3WoO.js';
|
|
53
53
|
import { a as TrajectoryStep, T as Trajectory } from './trajectory-2TkpSEVh.js';
|
|
@@ -59,28 +59,28 @@ export { a as CostChannel, c as CostLedgerEntry, d as CostLedgerSummary, e as Co
|
|
|
59
59
|
export { D as Direction, O as Objective, P as ParetoResult, c as crowdingDistance, d as dominates, p as paretoFrontier, a as paretoFrontierWithCrowding, s as scalarScore } from './pareto-E-pembql.js';
|
|
60
60
|
export { S as SeriesConvergenceOptions, a as SeriesConvergenceResult, b as analyzeSeries } from './series-convergence-D5OWMBg6.js';
|
|
61
61
|
import { D as DefaultVerdict } from './verdict-C9MlYujm.js';
|
|
62
|
-
import { a as DatasetScenario, b as Dataset } from './dataset-
|
|
63
|
-
export { d as DatasetDifficulty, c as DatasetManifest, e as DatasetProvenance, D as DatasetSplit, H as HoldoutLockedError, S as SliceOptions, h as hashScenarios } from './dataset-
|
|
62
|
+
import { a as DatasetScenario, b as Dataset } from './dataset-DS7ytHZU.js';
|
|
63
|
+
export { d as DatasetDifficulty, c as DatasetManifest, e as DatasetProvenance, D as DatasetSplit, H as HoldoutLockedError, S as SliceOptions, h as hashScenarios } from './dataset-DS7ytHZU.js';
|
|
64
64
|
export { a as CalibrationResult, c as CandidateScore, C as ContinuousAgreement, d as ContinuousAgreementOptions, b as ContinuousCalibrationResult, G as GoldenItem, P as PositionalBiasResult, S as SelfPreferenceResult, V as VerbosityBiasResult, e as calibrateJudge, f as calibrateJudgeContinuous, g as continuousAgreement, p as positionalBias, s as selfPreference, v as verbosityBias } from './judge-calibration-7C-IDmKr.js';
|
|
65
|
-
export { D as DEFAULT_RED_TEAM_CORPUS, R as RedTeamCase, a as RedTeamCategory, b as RedTeamFinding, c as RedTeamPayload, d as RedTeamReport, r as redTeamDataset, e as redTeamReport, s as scoreRedTeamOutput, t as toolNamesForRun } from './red-team-
|
|
65
|
+
export { D as DEFAULT_RED_TEAM_CORPUS, R as RedTeamCase, a as RedTeamCategory, b as RedTeamFinding, c as RedTeamPayload, d as RedTeamReport, r as redTeamDataset, e as redTeamReport, s as scoreRedTeamOutput, t as toolNamesForRun } from './red-team-KmmiqBlY.js';
|
|
66
66
|
export { c as CounterfactualContext, C as CounterfactualMutation, d as CounterfactualResult, b as CounterfactualRunner, a as attributeCounterfactuals, r as runCounterfactual } from './counterfactual-DlOz8PBx.js';
|
|
67
67
|
import { a as PrmGrader } from './rubric-Cc6UHvUb.js';
|
|
68
68
|
export { EuRiskClass, GovernanceContext, GovernanceFinding, GovernanceReport, UseCaseSignals, classifyEuAiRisk, euAiActReport, nistAiRmfReport, renderMarkdown, soc2Report, summarize } from './governance/index.js';
|
|
69
69
|
import { b as Layer, S as Severity, L as LayerResult, c as VerifyContext } from './multi-layer-verifier-CI4jdX-q.js';
|
|
70
70
|
export { F as Finding, d as LayerStatus, M as MultiLayerVerifier, a as VerificationReport, V as VerifyOptions, g as gradeSemanticStatus } from './multi-layer-verifier-CI4jdX-q.js';
|
|
71
|
-
import { L as LlmClientOptions } from './llm-client-
|
|
72
|
-
export { d as LlmCallError, b as LlmCallRequest, c as LlmCallResult, e as LlmClient, f as LlmMessage, g as LlmRouteAssertionError, a as LlmRouteRequirements, h as LlmUsage, i as assertLlmRoute, j as backoffMs, k as callLlm, l as callLlmJson, m as isTransientLlmError, p as probeLlm, s as stripFencedJson } from './llm-client-
|
|
73
|
-
export { B as BENCHMARK_SPLIT_SEED, a as BenchmarkAdapter, b as BenchmarkDatasetItem, c as BenchmarkEvaluation, d as benchmarkDeterministicSplit, i as benchmarks } from './index-
|
|
74
|
-
export { C as CallbackResearcher, d as CallbackResearcherOptions, e as CampaignFactoryParams, f as CampaignIntegrityPolicy, g as CampaignRunContext, h as CampaignRunOutcome, i as CampaignRunner, j as CampaignScenario, k as CampaignVariant, c as EvalCampaignOptions, b as EvalCampaignResult, E as ExperimentPlan, a as ExperimentResult, l as FailedRun, F as FailureMode, N as NoopResearcher, R as Researcher, S as SteeringChange, r as runEvalCampaign } from './researcher-
|
|
75
|
-
export { G as GainDistributionBin, a as GainDistributionFigureSpec, b as GainDistributionOptions, m as GateDecision, n as GateEvidence, H as HeldOutGate, o as HeldOutGateConfig, q as HeldOutGateRejectionCode, P as ParetoFigureSpec, c as ParetoPoint, R as RESEARCH_REPORT_HARD_PAIR_FLOOR, d as ResearchReport, e as ResearchReportCandidate, f as ResearchReportDecision, g as ResearchReportMethodology, h as ResearchReportOptions, i as ResearchReportRecommendation, S as SummaryTable, j as SummaryTableOptions, k as SummaryTableRow, l as gainHistogram, p as paretoChart, r as researchReport, s as summaryTable } from './summary-report-
|
|
71
|
+
import { L as LlmClientOptions } from './llm-client-DyqEH4jH.js';
|
|
72
|
+
export { d as LlmCallError, b as LlmCallRequest, c as LlmCallResult, e as LlmClient, f as LlmMessage, g as LlmRouteAssertionError, a as LlmRouteRequirements, h as LlmUsage, i as assertLlmRoute, j as backoffMs, k as callLlm, l as callLlmJson, m as isTransientLlmError, p as probeLlm, s as stripFencedJson } from './llm-client-DyqEH4jH.js';
|
|
73
|
+
export { B as BENCHMARK_SPLIT_SEED, a as BenchmarkAdapter, b as BenchmarkDatasetItem, c as BenchmarkEvaluation, d as benchmarkDeterministicSplit, i as benchmarks } from './index-C2CC_dry.js';
|
|
74
|
+
export { C as CallbackResearcher, d as CallbackResearcherOptions, e as CampaignFactoryParams, f as CampaignIntegrityPolicy, g as CampaignRunContext, h as CampaignRunOutcome, i as CampaignRunner, j as CampaignScenario, k as CampaignVariant, c as EvalCampaignOptions, b as EvalCampaignResult, E as ExperimentPlan, a as ExperimentResult, l as FailedRun, F as FailureMode, N as NoopResearcher, R as Researcher, S as SteeringChange, r as runEvalCampaign } from './researcher-Wc7dx6GM.js';
|
|
75
|
+
export { G as GainDistributionBin, a as GainDistributionFigureSpec, b as GainDistributionOptions, m as GateDecision, n as GateEvidence, H as HeldOutGate, o as HeldOutGateConfig, q as HeldOutGateRejectionCode, P as ParetoFigureSpec, c as ParetoPoint, R as RESEARCH_REPORT_HARD_PAIR_FLOOR, d as ResearchReport, e as ResearchReportCandidate, f as ResearchReportDecision, g as ResearchReportMethodology, h as ResearchReportOptions, i as ResearchReportRecommendation, S as SummaryTable, j as SummaryTableOptions, k as SummaryTableRow, l as gainHistogram, p as paretoChart, r as researchReport, s as summaryTable } from './summary-report-QMZVe3P-.js';
|
|
76
76
|
export { L as LockedJsonlAppender } from './testing-C21CHsq2.js';
|
|
77
77
|
export { I as InterimReleaseConfidence, a as InterimReleaseConfidenceInput, P as PairedEvalueOptions, b as PairedEvalueSequence, c as PairedEvalueStep, S as SequentialDecision, e as evaluateInterimReleaseConfidence, p as pairedEvalueSequence } from './sequential-5iSVfzl2.js';
|
|
78
|
-
import { j as GepaProposerConstraints, b as RunImprovementLoopResult } from './gepa-
|
|
78
|
+
import { j as GepaProposerConstraints, b as RunImprovementLoopResult } from './gepa-Dn0gtj9n.js';
|
|
79
79
|
export { IntegrityResult, IntegrityViolation, JourneySpec, PerfBaseline, PerfGateResult, PerfRegression, PerfScenario, PerfStat, ScenarioAxes, assertRecordIntegrity, checkRecordIntegrity, expandMatrix, gatePerf, scenarioKey, summarizeRecords } from './perf/index.js';
|
|
80
80
|
export { AgentProfileRuntimeReceipt, ProductBenchmarkArm, ProductBenchmarkArtifactPaths, ProductBenchmarkBudgets, ProductBenchmarkExportOptions, ProductBenchmarkExportResult, ProductBenchmarkManifest, ProductBenchmarkProfileRef, ProductBenchmarkRecord, ProductBenchmarkRepoRef, ProductBenchmarkRunInput, ProductBenchmarkScenario, ProductBenchmarkSingleRunExportOptions, ProductBenchmarkSplit, ProductBenchmarkSubstrateVersions, ProductBenchmarkValidationReport, RuntimeResolution, assertProductBenchmarkRun, buildProductBenchmarkManifest, exportProductBenchmark, exportProductBenchmarkRuns, findProductBenchmarkArtifacts, productBenchmarkIntegrityFailures, productBenchmarkMutableSurfaces, productBenchmarkRepoIdentity, productBenchmarkSplits, readProductBenchmarkManifest, readProductBenchmarkRecords, runRecordToProductBenchmarkRecord, validateProductBenchmarkManifest, validateProductBenchmarkRecord, validateProductBenchmarkRun } from './product-benchmark/index.js';
|
|
81
81
|
import '@ax-llm/ax';
|
|
82
82
|
import 'zod';
|
|
83
|
-
import './insight-report-
|
|
83
|
+
import './insight-report-B4xrdwEK.js';
|
|
84
84
|
import './outcome-store-rnXLEqSn.js';
|
|
85
85
|
|
|
86
86
|
/**
|
package/dist/index.js
CHANGED
|
@@ -9,7 +9,7 @@ import {
|
|
|
9
9
|
checkBehavioralCanary,
|
|
10
10
|
checkCanaries,
|
|
11
11
|
runBehavioralCanaries
|
|
12
|
-
} from "./chunk-
|
|
12
|
+
} from "./chunk-GDZAWO2I.js";
|
|
13
13
|
import {
|
|
14
14
|
classifyEuAiRisk,
|
|
15
15
|
euAiActReport,
|
|
@@ -49,7 +49,7 @@ import {
|
|
|
49
49
|
validateProductBenchmarkManifest,
|
|
50
50
|
validateProductBenchmarkRecord,
|
|
51
51
|
validateProductBenchmarkRun
|
|
52
|
-
} from "./chunk-
|
|
52
|
+
} from "./chunk-FOUG2VVS.js";
|
|
53
53
|
import {
|
|
54
54
|
CODING_HARNESSES,
|
|
55
55
|
HARNESS_NATIVE_MODEL,
|
|
@@ -72,7 +72,7 @@ import {
|
|
|
72
72
|
llmJudge,
|
|
73
73
|
parseCorrectnessResponse,
|
|
74
74
|
verifyCompletion
|
|
75
|
-
} from "./chunk-
|
|
75
|
+
} from "./chunk-BBQILVBH.js";
|
|
76
76
|
import {
|
|
77
77
|
DEFAULT_MUTATION_PRIMITIVES,
|
|
78
78
|
DEFAULT_RED_TEAM_CORPUS,
|
|
@@ -96,7 +96,7 @@ import {
|
|
|
96
96
|
scoreRedTeamOutput,
|
|
97
97
|
surfaceContentHash,
|
|
98
98
|
toolNamesForRun
|
|
99
|
-
} from "./chunk-
|
|
99
|
+
} from "./chunk-ZKIL3VOD.js";
|
|
100
100
|
import {
|
|
101
101
|
BackendIntegrityError,
|
|
102
102
|
assertRealBackend,
|
|
@@ -106,7 +106,7 @@ import {
|
|
|
106
106
|
fileVerdictCache,
|
|
107
107
|
inMemoryVerdictCache,
|
|
108
108
|
summarizeBackendIntegrity
|
|
109
|
-
} from "./chunk-
|
|
109
|
+
} from "./chunk-3PFZBGMR.js";
|
|
110
110
|
import {
|
|
111
111
|
MODEL_PRICING,
|
|
112
112
|
MetricsCollector,
|
|
@@ -128,7 +128,7 @@ import {
|
|
|
128
128
|
computeToolUseMetrics,
|
|
129
129
|
iqr,
|
|
130
130
|
welchsTTest
|
|
131
|
-
} from "./chunk-
|
|
131
|
+
} from "./chunk-DBDRR6GF.js";
|
|
132
132
|
import {
|
|
133
133
|
analyzeSeries
|
|
134
134
|
} from "./chunk-BOD4O7OF.js";
|
|
@@ -145,7 +145,7 @@ import {
|
|
|
145
145
|
pytestTestParser,
|
|
146
146
|
runTestGradedScenario,
|
|
147
147
|
vitestTestParser
|
|
148
|
-
} from "./chunk-
|
|
148
|
+
} from "./chunk-HZHNRYHK.js";
|
|
149
149
|
import {
|
|
150
150
|
DEFAULT_COMPLEXITY_WEIGHTS,
|
|
151
151
|
FindingsStore,
|
|
@@ -159,7 +159,7 @@ import {
|
|
|
159
159
|
defaultIsMaterial,
|
|
160
160
|
diffFindings,
|
|
161
161
|
runSemanticConceptJudge
|
|
162
|
-
} from "./chunk-
|
|
162
|
+
} from "./chunk-5PK3626Q.js";
|
|
163
163
|
import {
|
|
164
164
|
LockedJsonlAppender,
|
|
165
165
|
Mutex
|
|
@@ -167,7 +167,7 @@ import {
|
|
|
167
167
|
import {
|
|
168
168
|
buildDefaultAnalystRegistry,
|
|
169
169
|
computeTraceMetrics
|
|
170
|
-
} from "./chunk-
|
|
170
|
+
} from "./chunk-QRVS7MX4.js";
|
|
171
171
|
import {
|
|
172
172
|
DEFAULT_RUN_SCORE_WEIGHTS,
|
|
173
173
|
POLICY_EDIT_AXES,
|
|
@@ -184,7 +184,7 @@ import {
|
|
|
184
184
|
policyEditsFromFindings,
|
|
185
185
|
scorePolicyEditReadiness,
|
|
186
186
|
validatePolicyEdit
|
|
187
|
-
} from "./chunk-
|
|
187
|
+
} from "./chunk-LOJ2QVCE.js";
|
|
188
188
|
import {
|
|
189
189
|
AnalystRegistry,
|
|
190
190
|
DEFAULT_TRACE_ANALYST_KINDS,
|
|
@@ -194,7 +194,7 @@ import {
|
|
|
194
194
|
KNOWLEDGE_POISONING_KIND_SPEC,
|
|
195
195
|
createTraceAnalystKind,
|
|
196
196
|
renderPriorFindings
|
|
197
|
-
} from "./chunk-
|
|
197
|
+
} from "./chunk-2OGPXHOB.js";
|
|
198
198
|
import {
|
|
199
199
|
controlFailureClassFromVerification,
|
|
200
200
|
controlRunToRunRecord,
|
|
@@ -205,7 +205,7 @@ import {
|
|
|
205
205
|
runProposeReview,
|
|
206
206
|
runProposeReviewAsControlLoop,
|
|
207
207
|
scoreFromEvals
|
|
208
|
-
} from "./chunk-
|
|
208
|
+
} from "./chunk-K7QEIHHJ.js";
|
|
209
209
|
import {
|
|
210
210
|
allCriticalPassed,
|
|
211
211
|
errorStreakDetector,
|
|
@@ -224,10 +224,10 @@ import {
|
|
|
224
224
|
evaluateReleaseConfidence,
|
|
225
225
|
judgeReplayGate,
|
|
226
226
|
renderReleaseReport
|
|
227
|
-
} from "./chunk-
|
|
227
|
+
} from "./chunk-6HFYGEZ3.js";
|
|
228
228
|
import {
|
|
229
229
|
runEvalCampaign
|
|
230
|
-
} from "./chunk-
|
|
230
|
+
} from "./chunk-LIEJUH2I.js";
|
|
231
231
|
import {
|
|
232
232
|
LlmCallError,
|
|
233
233
|
LlmClient,
|
|
@@ -239,7 +239,7 @@ import {
|
|
|
239
239
|
isTransientLlmError,
|
|
240
240
|
probeLlm,
|
|
241
241
|
stripFencedJson
|
|
242
|
-
} from "./chunk-
|
|
242
|
+
} from "./chunk-FUCQVFMU.js";
|
|
243
243
|
import {
|
|
244
244
|
evaluateInterimReleaseConfidence,
|
|
245
245
|
pairedEvalueSequence
|
|
@@ -250,11 +250,11 @@ import {
|
|
|
250
250
|
paretoChart,
|
|
251
251
|
researchReport,
|
|
252
252
|
summaryTable
|
|
253
|
-
} from "./chunk-
|
|
253
|
+
} from "./chunk-6MFFKPZ4.js";
|
|
254
254
|
import {
|
|
255
255
|
attributeCounterfactuals,
|
|
256
256
|
runCounterfactual
|
|
257
|
-
} from "./chunk-
|
|
257
|
+
} from "./chunk-6SK5VFYK.js";
|
|
258
258
|
import {
|
|
259
259
|
buildTrajectory
|
|
260
260
|
} from "./chunk-RZTMDUO7.js";
|
|
@@ -299,7 +299,7 @@ import {
|
|
|
299
299
|
weightedMean,
|
|
300
300
|
wilcoxonSignedRank,
|
|
301
301
|
wilson
|
|
302
|
-
} from "./chunk-
|
|
302
|
+
} from "./chunk-XMSYF4A7.js";
|
|
303
303
|
import {
|
|
304
304
|
FileSystemTraceStore,
|
|
305
305
|
InMemoryTraceStore,
|
|
@@ -330,13 +330,13 @@ import {
|
|
|
330
330
|
scoreTraceInsightReadiness,
|
|
331
331
|
tokenizeDomainWords,
|
|
332
332
|
traceAnalystOnRunComplete
|
|
333
|
-
} from "./chunk-
|
|
333
|
+
} from "./chunk-V7HNA47Z.js";
|
|
334
334
|
import {
|
|
335
335
|
TRACE_ANALYST_ACTOR_DESCRIPTION,
|
|
336
336
|
TRACE_ANALYST_ACTOR_DESCRIPTION_VERSION,
|
|
337
337
|
TRACE_ANALYST_SUBAGENT_DESCRIPTION,
|
|
338
338
|
analyzeTraces
|
|
339
|
-
} from "./chunk-
|
|
339
|
+
} from "./chunk-RPDDVKI7.js";
|
|
340
340
|
import {
|
|
341
341
|
DEFAULT_REDACTION_RULES,
|
|
342
342
|
REDACTION_VERSION,
|
|
@@ -395,12 +395,12 @@ import {
|
|
|
395
395
|
stringField,
|
|
396
396
|
traceAnalystFunctionGroup,
|
|
397
397
|
traceSpanKindToOpenInferenceKind
|
|
398
|
-
} from "./chunk-
|
|
398
|
+
} from "./chunk-LNQEP766.js";
|
|
399
399
|
import {
|
|
400
400
|
RunIntegrityError,
|
|
401
401
|
assertRunCaptured,
|
|
402
402
|
throwIfRunIncomplete
|
|
403
|
-
} from "./chunk-
|
|
403
|
+
} from "./chunk-TT4KNT67.js";
|
|
404
404
|
import {
|
|
405
405
|
FileSystemRawProviderSink,
|
|
406
406
|
InMemoryRawProviderSink,
|
|
@@ -415,7 +415,7 @@ import {
|
|
|
415
415
|
parseRunRecordSafe,
|
|
416
416
|
roundTripRunRecord,
|
|
417
417
|
validateRunRecord
|
|
418
|
-
} from "./chunk-
|
|
418
|
+
} from "./chunk-VK6HBGAE.js";
|
|
419
419
|
import {
|
|
420
420
|
TraceEmitter,
|
|
421
421
|
llmSpanFromProvider
|
|
@@ -433,7 +433,7 @@ import {
|
|
|
433
433
|
toAgentProfileJson,
|
|
434
434
|
validateAgentProfileCell,
|
|
435
435
|
verifyAgentProfileCell
|
|
436
|
-
} from "./chunk-
|
|
436
|
+
} from "./chunk-XJYR7XFV.js";
|
|
437
437
|
import {
|
|
438
438
|
canonicalize,
|
|
439
439
|
evaluateHypothesis,
|
|
@@ -450,7 +450,7 @@ import {
|
|
|
450
450
|
ReplayError,
|
|
451
451
|
ValidationError,
|
|
452
452
|
VerificationError
|
|
453
|
-
} from "./chunk-
|
|
453
|
+
} from "./chunk-ONWEPEDO.js";
|
|
454
454
|
import {
|
|
455
455
|
__export
|
|
456
456
|
} from "./chunk-PZ5AY32C.js";
|
|
@@ -1,4 +1,4 @@
|
|
|
1
|
-
import { G as GainDistributionBin, P as ParetoFigureSpec } from './summary-report-
|
|
1
|
+
import { G as GainDistributionBin, P as ParetoFigureSpec } from './summary-report-QMZVe3P-.js';
|
|
2
2
|
import { C as ContinuousAgreement } from './judge-calibration-7C-IDmKr.js';
|
|
3
3
|
|
|
4
4
|
/**
|
|
@@ -1,7 +1,7 @@
|
|
|
1
1
|
import { AxAIService, AxFunction } from '@ax-llm/ax';
|
|
2
2
|
import { T as TraceAnalysisStore } from './store-C1YxJDEK.js';
|
|
3
3
|
import { z } from 'zod';
|
|
4
|
-
import { g as AnalystCost, b as AnalystContext, a as Analyst } from './types-
|
|
4
|
+
import { g as AnalystCost, b as AnalystContext, a as Analyst } from './types-D1ytG0Yg.js';
|
|
5
5
|
|
|
6
6
|
/**
|
|
7
7
|
* Typed Ax output for analyst findings.
|
|
@@ -1,4 +1,4 @@
|
|
|
1
|
-
import { A as AgentEvalError, C as CaptureIntegrityError } from './errors-
|
|
1
|
+
import { A as AgentEvalError, C as CaptureIntegrityError } from './errors-oeQrLqXC.js';
|
|
2
2
|
import { R as RawProviderSink, P as ProviderRedactor } from './raw-provider-sink-C46HDghv.js';
|
|
3
3
|
|
|
4
4
|
/**
|
|
@@ -1,14 +1,14 @@
|
|
|
1
1
|
export { C as CalibrationBin, a as CalibrationOptions, b as CalibrationPair, c as CalibrationReport, d as CorrelationResult, e as CorrelationStudyOptions, f as CorrelationStudyResult, E as EvalMetricSpec, O as OutcomePair, g as calibrationCurve, h as calibrationFromPairs, i as correlationStudy } from '../calibration-BPmzuVPk.js';
|
|
2
2
|
export { D as DeploymentOutcome, F as FileSystemOutcomeStore, a as FileSystemOutcomeStoreOptions, I as InMemoryOutcomeStore, O as OutcomeFilter, b as OutcomeStore } from '../outcome-store-rnXLEqSn.js';
|
|
3
|
-
export { R as RubricOutcomePair, a as RubricPredictiveValidityInput, b as RubricPredictiveValidityReport, c as RubricRanking, r as rubricPredictiveValidity } from '../rubric-predictive-validity-
|
|
3
|
+
export { R as RubricOutcomePair, a as RubricPredictiveValidityInput, b as RubricPredictiveValidityReport, c as RubricRanking, r as rubricPredictiveValidity } from '../rubric-predictive-validity-DPnyG-CE.js';
|
|
4
4
|
import { C as ContinuousAgreement, a as CalibrationResult, b as ContinuousCalibrationResult, c as CandidateScore, G as GoldenItem } from '../judge-calibration-7C-IDmKr.js';
|
|
5
5
|
import { S as SeriesConvergenceOptions, a as SeriesConvergenceResult } from '../series-convergence-D5OWMBg6.js';
|
|
6
6
|
import { C as CorpusAgreementReport } from '../statistics-D88peojY.js';
|
|
7
7
|
import '../store-BcFXE6LG.js';
|
|
8
8
|
import '../schema-m0gsnbt3.js';
|
|
9
|
-
import '../run-record-
|
|
9
|
+
import '../run-record-I-Z3JNvO.js';
|
|
10
10
|
import '@tangle-network/agent-interface';
|
|
11
|
-
import '../errors-
|
|
11
|
+
import '../errors-oeQrLqXC.js';
|
|
12
12
|
import '../types-C7DGg5ex.js';
|
|
13
13
|
import '@tangle-network/tcloud';
|
|
14
14
|
|
package/dist/meta-eval/index.js
CHANGED
|
@@ -11,11 +11,11 @@ import {
|
|
|
11
11
|
} from "../chunk-3RF76KTD.js";
|
|
12
12
|
import {
|
|
13
13
|
rubricPredictiveValidity
|
|
14
|
-
} from "../chunk-
|
|
14
|
+
} from "../chunk-X6NIXVOD.js";
|
|
15
15
|
import {
|
|
16
16
|
pearsonR,
|
|
17
17
|
spearmanR
|
|
18
|
-
} from "../chunk-
|
|
18
|
+
} from "../chunk-XMSYF4A7.js";
|
|
19
19
|
import {
|
|
20
20
|
aggregateLlm,
|
|
21
21
|
llmSpans
|
|
@@ -23,7 +23,7 @@ import {
|
|
|
23
23
|
import "../chunk-5BKGXME7.js";
|
|
24
24
|
import {
|
|
25
25
|
ValidationError
|
|
26
|
-
} from "../chunk-
|
|
26
|
+
} from "../chunk-ONWEPEDO.js";
|
|
27
27
|
import "../chunk-PZ5AY32C.js";
|
|
28
28
|
|
|
29
29
|
// src/meta-eval/correlation-study.ts
|
|
@@ -1,8 +1,8 @@
|
|
|
1
|
-
import { J as JudgeScore } from '../types-
|
|
1
|
+
import { J as JudgeScore } from '../types-Mh6I44ub.js';
|
|
2
2
|
import { AgentProfile } from '@tangle-network/agent-interface';
|
|
3
3
|
import { M as MatrixResult } from '../types-BUxNaJ8c.js';
|
|
4
|
-
import '../run-record-
|
|
5
|
-
import '../errors-
|
|
4
|
+
import '../run-record-I-Z3JNvO.js';
|
|
5
|
+
import '../errors-oeQrLqXC.js';
|
|
6
6
|
import '../schema-m0gsnbt3.js';
|
|
7
7
|
import '../verdict-C9MlYujm.js';
|
|
8
8
|
|
|
@@ -40,6 +40,45 @@ interface MultishotToolDefinition {
|
|
|
40
40
|
parameters: Record<string, unknown>;
|
|
41
41
|
};
|
|
42
42
|
}
|
|
43
|
+
/** One chat-completion request the multishot loop issues for a single agent
|
|
44
|
+
* (or driver) inference step. Mirrors the OpenAI-compat body the loop would
|
|
45
|
+
* otherwise POST to the Tangle router. */
|
|
46
|
+
interface MultishotTransportRequest {
|
|
47
|
+
model: string;
|
|
48
|
+
messages: Array<Record<string, unknown>>;
|
|
49
|
+
tools?: MultishotToolDefinition[];
|
|
50
|
+
temperature?: number;
|
|
51
|
+
maxTokens?: number;
|
|
52
|
+
signal?: AbortSignal;
|
|
53
|
+
}
|
|
54
|
+
interface MultishotTransportToolCall {
|
|
55
|
+
id: string;
|
|
56
|
+
type: 'function';
|
|
57
|
+
function: {
|
|
58
|
+
name: string;
|
|
59
|
+
arguments: string;
|
|
60
|
+
};
|
|
61
|
+
}
|
|
62
|
+
interface MultishotTransportResponse {
|
|
63
|
+
message: {
|
|
64
|
+
content?: string | null;
|
|
65
|
+
tool_calls?: MultishotTransportToolCall[];
|
|
66
|
+
};
|
|
67
|
+
usage?: {
|
|
68
|
+
prompt_tokens?: number;
|
|
69
|
+
completion_tokens?: number;
|
|
70
|
+
};
|
|
71
|
+
/** Actual spend for this call. When omitted, the loop meters cost from
|
|
72
|
+
* `usage` via the per-model router estimator (estimateRouterCost). */
|
|
73
|
+
costUsd?: number;
|
|
74
|
+
}
|
|
75
|
+
/** Execution seam for one leg of the multishot loop. When provided, it
|
|
76
|
+
* replaces the internal router HTTP call for that leg — the loop still owns
|
|
77
|
+
* turn scheduling, tool dispatch, transcript capture, and cost metering.
|
|
78
|
+
* agent-eval has no dependency on agent-runtime; adapt agent-runtime's
|
|
79
|
+
* resolveAgentBackend (or any sandbox/cli-bridge/router client) into this
|
|
80
|
+
* signature product-side. */
|
|
81
|
+
type MultishotTransport = (req: MultishotTransportRequest) => Promise<MultishotTransportResponse>;
|
|
43
82
|
type MultishotToolExecutor = (args: Record<string, unknown>, ctx: {
|
|
44
83
|
apiKey: string;
|
|
45
84
|
baseUrl: string;
|
|
@@ -251,6 +290,12 @@ interface RunMultishotMatrixOptions<TPersona extends MultishotPersona> {
|
|
|
251
290
|
driverMaxTokens?: number;
|
|
252
291
|
/** Maximum output tokens for each judge response. */
|
|
253
292
|
judgeMaxTokens?: number;
|
|
293
|
+
/** Execution seam for the agent leg of every cell — replaces the router
|
|
294
|
+
* HTTP call when provided (see RunMultishotOptions.agentTransport).
|
|
295
|
+
* Judges are unaffected; configure those via MultishotJudges. */
|
|
296
|
+
agentTransport?: MultishotTransport;
|
|
297
|
+
/** Execution seam for the simulated-user driver leg of every cell. */
|
|
298
|
+
driverTransport?: MultishotTransport;
|
|
254
299
|
/** Pass-thru fields. */
|
|
255
300
|
apiKey?: string;
|
|
256
301
|
baseUrl?: string;
|
|
@@ -308,10 +353,20 @@ interface RunMultishotOptions<TPersona extends MultishotPersona> {
|
|
|
308
353
|
driverMaxTokens?: number;
|
|
309
354
|
/** Maximum tool calls the agent may dispatch inside one assistant turn. */
|
|
310
355
|
maxToolDispatches?: number;
|
|
356
|
+
/** Execution seam for the agent leg. When provided, every agent inference
|
|
357
|
+
* step goes through this function instead of the router HTTP call; the
|
|
358
|
+
* string levers (agentModel, apiKey, baseUrl) stop applying to that leg.
|
|
359
|
+
* apiKey/baseUrl are still resolved for tool executors and any leg
|
|
360
|
+
* without an injected transport. */
|
|
361
|
+
agentTransport?: MultishotTransport;
|
|
362
|
+
/** Execution seam for the simulated-user driver leg (symmetric to
|
|
363
|
+
* agentTransport). Driver model fallback rotation still applies — the
|
|
364
|
+
* transport receives each candidate model in turn. */
|
|
365
|
+
driverTransport?: MultishotTransport;
|
|
311
366
|
apiKey?: string;
|
|
312
367
|
baseUrl?: string;
|
|
313
368
|
signal?: AbortSignal;
|
|
314
369
|
}
|
|
315
370
|
declare function runMultishot<TPersona extends MultishotPersona>(opts: RunMultishotOptions<TPersona>): Promise<MultishotResult>;
|
|
316
371
|
|
|
317
|
-
export { type ArtifactJudgeInput, type CellCompositeInput, type CellCompositeScore, type ConversationJudgeInput, DEFAULT_CODER_MODEL, DEFAULT_DELEGATE_CODE_TOOL, DEFAULT_DELEGATE_RESEARCH_TOOL, DEFAULT_JUDGE_MODEL, DEFAULT_RESEARCHER_MODEL, type DefaultCoderConfig, type DefaultResearcherConfig, type DefaultToolsBundle, type DefaultToolsConfig, type JudgeConfig, type JudgeDimension, JudgeScore, type MultishotArtifact, MultishotDriverEmptyError, MultishotFatalToolError, type MultishotJudges, type MultishotMessage, type MultishotPersona, type MultishotResult, type MultishotShape, type MultishotToolDefinition, type MultishotToolExecutor, type RouterCompletionRequest, type RouterCompletionResponse, type RouterToolCall, type RunMultishotMatrixOptions, type RunMultishotMatrixResult, type RunMultishotOptions, computeCellComposite, createCodeExecutor, createResearchExecutor, defaultDelegationTools, defaultRouterBaseUrl, estimateRouterCost, renderDimensions, renderJsonFooter, requireRouterApiKey, routerCompletion, runJudge, runMultishot, runMultishotMatrix };
|
|
372
|
+
export { type ArtifactJudgeInput, type CellCompositeInput, type CellCompositeScore, type ConversationJudgeInput, DEFAULT_CODER_MODEL, DEFAULT_DELEGATE_CODE_TOOL, DEFAULT_DELEGATE_RESEARCH_TOOL, DEFAULT_JUDGE_MODEL, DEFAULT_RESEARCHER_MODEL, type DefaultCoderConfig, type DefaultResearcherConfig, type DefaultToolsBundle, type DefaultToolsConfig, type JudgeConfig, type JudgeDimension, JudgeScore, type MultishotArtifact, MultishotDriverEmptyError, MultishotFatalToolError, type MultishotJudges, type MultishotMessage, type MultishotPersona, type MultishotResult, type MultishotShape, type MultishotToolDefinition, type MultishotToolExecutor, type MultishotTransport, type MultishotTransportRequest, type MultishotTransportResponse, type MultishotTransportToolCall, type RouterCompletionRequest, type RouterCompletionResponse, type RouterToolCall, type RunMultishotMatrixOptions, type RunMultishotMatrixResult, type RunMultishotOptions, computeCellComposite, createCodeExecutor, createResearchExecutor, defaultDelegationTools, defaultRouterBaseUrl, estimateRouterCost, renderDimensions, renderJsonFooter, requireRouterApiKey, routerCompletion, runJudge, runMultishot, runMultishotMatrix };
|
package/dist/multishot/index.js
CHANGED
|
@@ -259,6 +259,9 @@ async function runMultishot(opts) {
|
|
|
259
259
|
const tools = opts.tools ?? bundle.tools;
|
|
260
260
|
const executors = opts.toolExecutors ?? bundle.executors;
|
|
261
261
|
const artifactTypeFor = opts.artifactTypeFor ?? bundle.artifactTypeFor;
|
|
262
|
+
const routerTransport = (req) => routerCompletion({ apiKey, baseUrl, ...req });
|
|
263
|
+
const agentTransport = opts.agentTransport ?? routerTransport;
|
|
264
|
+
const driverTransport = opts.driverTransport ?? routerTransport;
|
|
262
265
|
const start = Date.now();
|
|
263
266
|
const transcript = [];
|
|
264
267
|
const artifacts = [];
|
|
@@ -275,9 +278,11 @@ async function runMultishot(opts) {
|
|
|
275
278
|
if (opts.signal?.aborted) throw new Error("multishot aborted");
|
|
276
279
|
let dispatchesThisTurn = 0;
|
|
277
280
|
while (true) {
|
|
278
|
-
const {
|
|
279
|
-
|
|
280
|
-
|
|
281
|
+
const {
|
|
282
|
+
message: agentMsg,
|
|
283
|
+
usage: agentUsage,
|
|
284
|
+
costUsd: agentCostUsd
|
|
285
|
+
} = await agentTransport({
|
|
281
286
|
model: agentModel,
|
|
282
287
|
messages: agentMessages,
|
|
283
288
|
tools,
|
|
@@ -285,7 +290,7 @@ async function runMultishot(opts) {
|
|
|
285
290
|
maxTokens: dispatchesThisTurn === 0 ? agentMaxTokens : toolFollowupMaxTokens,
|
|
286
291
|
signal: opts.signal
|
|
287
292
|
});
|
|
288
|
-
totalCostUsd += estimateRouterCost(agentModel, agentUsage);
|
|
293
|
+
totalCostUsd += agentCostUsd ?? estimateRouterCost(agentModel, agentUsage);
|
|
289
294
|
const agentText = (agentMsg.content ?? "").trim();
|
|
290
295
|
const agentToolCalls = (agentMsg.tool_calls ?? []).map((tc) => ({
|
|
291
296
|
id: tc.id,
|
|
@@ -346,8 +351,7 @@ async function runMultishot(opts) {
|
|
|
346
351
|
}
|
|
347
352
|
if (turn < maxTurns - 1) {
|
|
348
353
|
const driver = await driverTurn({
|
|
349
|
-
|
|
350
|
-
baseUrl,
|
|
354
|
+
transport: driverTransport,
|
|
351
355
|
persona: opts.persona,
|
|
352
356
|
shape: opts.shape,
|
|
353
357
|
transcript,
|
|
@@ -375,9 +379,7 @@ async function driverTurn(opts) {
|
|
|
375
379
|
}
|
|
376
380
|
for (const model of opts.models) {
|
|
377
381
|
for (let attempt = 0; attempt < 2; attempt++) {
|
|
378
|
-
const { message, usage } = await
|
|
379
|
-
apiKey: opts.apiKey,
|
|
380
|
-
baseUrl: opts.baseUrl,
|
|
382
|
+
const { message, usage, costUsd } = await opts.transport({
|
|
381
383
|
model,
|
|
382
384
|
messages: driverMessages,
|
|
383
385
|
temperature: 0.9,
|
|
@@ -385,7 +387,8 @@ async function driverTurn(opts) {
|
|
|
385
387
|
signal: opts.signal
|
|
386
388
|
});
|
|
387
389
|
const content = (message.content ?? "").trim();
|
|
388
|
-
if (content.length > 0)
|
|
390
|
+
if (content.length > 0)
|
|
391
|
+
return { content, costUsd: costUsd ?? estimateRouterCost(model, usage) };
|
|
389
392
|
}
|
|
390
393
|
}
|
|
391
394
|
throw new MultishotDriverEmptyError(opts.turn);
|
|
@@ -452,6 +455,8 @@ async function runMultishotMatrix(opts) {
|
|
|
452
455
|
agentMaxTokens: opts.agentMaxTokens,
|
|
453
456
|
toolFollowupMaxTokens: opts.toolFollowupMaxTokens,
|
|
454
457
|
driverMaxTokens: opts.driverMaxTokens,
|
|
458
|
+
agentTransport: opts.agentTransport,
|
|
459
|
+
driverTransport: opts.driverTransport,
|
|
455
460
|
apiKey: opts.apiKey,
|
|
456
461
|
baseUrl: opts.baseUrl
|
|
457
462
|
});
|