@tangle-network/agent-eval 0.115.0 → 0.115.2
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +20 -0
- package/dist/analyst/index.d.ts +7 -7
- package/dist/{analyze-runs-Dmz6LA9e.d.ts → analyze-runs-BYHg6Irm.d.ts} +3 -3
- package/dist/belief-state/index.d.ts +3 -3
- package/dist/belief-state/index.js +1 -1
- package/dist/benchmarks/index.d.ts +3 -3
- package/dist/benchmarks/index.js +2 -2
- package/dist/campaign/index.d.ts +13 -12
- package/dist/campaign/index.js +2 -2
- package/dist/{chunk-MNR6ZW4P.js → chunk-5NVBGKPH.js} +3 -3
- package/dist/chunk-5NVBGKPH.js.map +1 -0
- package/dist/{chunk-VK6HBGAE.js → chunk-5UF54T55.js} +53 -1
- package/dist/chunk-5UF54T55.js.map +1 -0
- package/dist/{chunk-IMWDSFUM.js → chunk-DXZRATT5.js} +2 -2
- package/dist/{chunk-S42AWHMP.js → chunk-E4BUPP7Z.js} +135 -17
- package/dist/chunk-E4BUPP7Z.js.map +1 -0
- package/dist/{chunk-WBOGKYM4.js → chunk-J6P6PK2R.js} +48 -7
- package/dist/chunk-J6P6PK2R.js.map +1 -0
- package/dist/{chunk-LOBMT6SB.js → chunk-QG5F6463.js} +2 -2
- package/dist/{chunk-RSVSSZKF.js → chunk-TLDB7WRY.js} +2 -2
- package/dist/{code-agent-session-yitf9I-F.d.ts → code-agent-session-D-g04tcy.d.ts} +8 -1
- package/dist/contract/index.d.ts +15 -15
- package/dist/contract/index.js +3 -3
- package/dist/{control-U8LBKUES.d.ts → control-CcBiAEnn.d.ts} +1 -1
- package/dist/control.d.ts +2 -2
- package/dist/control.js +2 -2
- package/dist/{default-registry-Bcf1uKVI.d.ts → default-registry-DltpYR5u.d.ts} +1 -1
- package/dist/{gepa-DolL_Fko.d.ts → gepa-dne9JDPL.d.ts} +1 -1
- package/dist/hosted/index.d.ts +4 -4
- package/dist/{index-CWr5SIG-.d.ts → index-BTEpx9He.d.ts} +2 -2
- package/dist/index.d.ts +22 -22
- package/dist/index.js +8 -6
- package/dist/index.js.map +1 -1
- package/dist/{insight-report-D4cXFsLt.d.ts → insight-report-IwwvqZZv.d.ts} +20 -2
- package/dist/{kind-factory-20hcaYpf.d.ts → kind-factory-DcNg13sZ.d.ts} +1 -1
- package/dist/meta-eval/index.d.ts +2 -2
- package/dist/multishot/index.d.ts +2 -2
- package/dist/openapi.json +1 -1
- package/dist/{policy-edit-az2qRmvN.d.ts → policy-edit-RLn8GWof.d.ts} +2 -2
- package/dist/{pre-registration-oNItiRBb.d.ts → pre-registration-D8h7ZxNL.d.ts} +3 -3
- package/dist/{provenance-BZmpWmn4.d.ts → provenance-Bibyg1U9.d.ts} +3 -3
- package/dist/{release-report-oBfOz8ku.d.ts → release-report-CCtzajxP.d.ts} +2 -2
- package/dist/reporting.d.ts +4 -4
- package/dist/{researcher-CaH0CwFC.d.ts → researcher-Dq-EtpbE.d.ts} +2 -2
- package/dist/rl.d.ts +6 -6
- package/dist/rl.js +2 -2
- package/dist/{rubric-predictive-validity-C-fMteAW.d.ts → rubric-predictive-validity-DYTLjGWu.d.ts} +1 -1
- package/dist/{run-record-DksGsfgv.d.ts → run-record-B7RTi_ix.d.ts} +34 -2
- package/dist/{runtime-trajectory-h5i0SZUj.d.ts → runtime-trajectory-Dws7Kpgi.d.ts} +1 -1
- package/dist/{semantic-concept-judge-CpzbtwD0.d.ts → semantic-concept-judge-DxJmRkyJ.d.ts} +1 -1
- package/dist/{summary-report-Bz-0-t8v.d.ts → summary-report-BJ5aNwZ1.d.ts} +1 -1
- package/dist/traces.d.ts +1 -1
- package/dist/traces.js +2 -2
- package/dist/{types-CgSlO6wT.d.ts → types-C5gJrOVT.d.ts} +1 -1
- package/package.json +1 -1
- package/dist/chunk-MNR6ZW4P.js.map +0 -1
- package/dist/chunk-S42AWHMP.js.map +0 -1
- package/dist/chunk-VK6HBGAE.js.map +0 -1
- package/dist/chunk-WBOGKYM4.js.map +0 -1
- /package/dist/{chunk-IMWDSFUM.js.map → chunk-DXZRATT5.js.map} +0 -0
- /package/dist/{chunk-LOBMT6SB.js.map → chunk-QG5F6463.js.map} +0 -0
- /package/dist/{chunk-RSVSSZKF.js.map → chunk-TLDB7WRY.js.map} +0 -0
package/dist/index.d.ts
CHANGED
|
@@ -1,12 +1,12 @@
|
|
|
1
|
-
export { A as ActionExecutionPolicy, a as ActionPolicyDecision, C as ControlRunToRunRecordOptions, L as LlmJsonCall, b as LlmReviewerConfig, P as ProposeFn, c as ProposeInput, d as ProposeOutput, e as ProposeReviewConfig, f as ProposeReviewControlAction, g as ProposeReviewControlConfig, h as ProposeReviewControlResult, i as ProposeReviewControlState, j as ProposeReviewReport, k as ProposeReviewShot, R as Review, l as ReviewFn, m as ReviewInput, n as ReviewMemoryEntry, o as ReviewMemoryStore, p as RunEvidenceMetadata, V as Verification, q as VerifyFn, r as controlFailureClassFromVerification, s as controlRunToRunRecord, t as createLlmReviewer, u as evaluateActionPolicy, v as inMemoryReviewStore, w as jsonlReviewStore, x as runProposeReview, y as runProposeReviewAsControlLoop, z as scoreFromEvals } from './control-
|
|
2
|
-
import { R as RunRecord, b as RunSplitTag } from './run-record-
|
|
3
|
-
export {
|
|
4
|
-
import { B as BehavioralMetrics, A as RunScore, a as RunTrace, E as RunScoreWeights } from './semantic-concept-judge-
|
|
5
|
-
export { G as ConceptComplexity, H as ConceptFinding, J as ConceptSpec, L as ConceptWeightStrategy, C as CreateAnalystAiConfig, M as DEFAULT_COMPLEXITY_WEIGHTS, N as DEFAULT_RUN_SCORE_WEIGHTS, D as DEFAULT_TRACE_ANALYST_KINDS, c as DiffPolicy, F as FAILURE_MODE_KIND_SPEC, g as FindingSubject, h as FindingSubjectKind, j as FindingsDiff, k as FindingsStore, I as IMPROVEMENT_KIND_SPEC, l as KNOWLEDGE_GAP_KIND_SPEC, m as KNOWLEDGE_POISONING_KIND_SPEC, P as PersistedFinding, R as RunCritic, O as RunCriticOptions, Q as SEMANTIC_CONCEPT_JUDGE_VERSION, n as SKILL_USAGE_ANALYST, b as SemanticConceptJudgeInput, S as SemanticConceptJudgeOptions, T as SemanticConceptJudgeResult, o as SkillUsageAnalyst, U as SuboptimalCode, V as SuboptimalSignal, W as aggregateRunScore, X as clamp01, Y as computeTraceMetrics, t as createAnalystAi, Z as createSemanticConceptJudge, u as defaultIsMaterial, v as diffFindings, _ as runSemanticConceptJudge } from './semantic-concept-judge-
|
|
6
|
-
import { m as ChatRequest, q as CreateChatClientOpts } from './kind-factory-
|
|
7
|
-
export { a as Analyst, b as AnalystContext, i as AnalystCost, A as AnalystFinding, j as AnalystInputKind, k as AnalystRequirements, f as AnalystRunEvent, e as AnalystRunInputs, d as AnalystRunResult, c as AnalystRunSummary, g as AnalystSeverity, l as ChatCallOpts, C as ChatClient, n as ChatResponse, o as ChatTransport, p as CliBridgeTransportOpts, r as CreateTraceAnalystKindOpts, D as DirectProviderTransportOpts, E as EvidenceRef, M as MockTransportOpts, s as RawAnalystFinding, u as RouterTransportOpts, S as SandboxSdkTransportOpts, v as TraceAnalystGolden, T as TraceAnalystKindSpec, w as computeFindingId, x as createChatClient, y as createTraceAnalystKind, z as makeFinding, F as renderPriorFindings } from './kind-factory-
|
|
8
|
-
export { a as AnalystHooks, A as AnalystRegistry, b as AnalystRegistryOptions, B as BudgetPolicy, D as DefaultAnalystRegistryOptions, R as RegistryRunOpts, c as buildDefaultAnalystRegistry } from './default-registry-
|
|
9
|
-
export { F as FindingToPolicyEditOptions, P as POLICY_EDIT_AXES, a as POLICY_EDIT_TARGET_SURFACES, b as PolicyEdit, c as PolicyEditAdmission, d as PolicyEditAdmissionOptions, e as PolicyEditAxis, f as PolicyEditChange, g as PolicyEditExpectedGain, h as PolicyEditGainDirection, i as PolicyEditGainUnit, j as PolicyEditInit, k as PolicyEditRisk, l as PolicyEditSchemaVersion, m as PolicyEditSource, n as PolicyEditTarget, o as PolicyEditTargetSurface, p as PolicyEditValidationError, q as admitPolicyEdit, r as applyPolicyEditToSurface, s as computePolicyEditId, t as isPolicyEdit, u as makePolicyEdit, v as policyEditFromFinding, w as policyEditsFromFindings, x as scorePolicyEditReadiness, y as validatePolicyEdit } from './policy-edit-
|
|
1
|
+
export { A as ActionExecutionPolicy, a as ActionPolicyDecision, C as ControlRunToRunRecordOptions, L as LlmJsonCall, b as LlmReviewerConfig, P as ProposeFn, c as ProposeInput, d as ProposeOutput, e as ProposeReviewConfig, f as ProposeReviewControlAction, g as ProposeReviewControlConfig, h as ProposeReviewControlResult, i as ProposeReviewControlState, j as ProposeReviewReport, k as ProposeReviewShot, R as Review, l as ReviewFn, m as ReviewInput, n as ReviewMemoryEntry, o as ReviewMemoryStore, p as RunEvidenceMetadata, V as Verification, q as VerifyFn, r as controlFailureClassFromVerification, s as controlRunToRunRecord, t as createLlmReviewer, u as evaluateActionPolicy, v as inMemoryReviewStore, w as jsonlReviewStore, x as runProposeReview, y as runProposeReviewAsControlLoop, z as scoreFromEvals } from './control-CcBiAEnn.js';
|
|
2
|
+
import { R as RunRecord, b as RunSplitTag } from './run-record-B7RTi_ix.js';
|
|
3
|
+
export { g as AGENT_PROFILE_KINDS, h as AgentInterfaceProfileLike, A as AgentProfileCell, f as AgentProfileCellInput, i as AgentProfileCellSchemaVersion, j as AgentProfileCellValidationError, k as AgentProfileDimensionValue, l as AgentProfileHarness, a as AgentProfileJson, m as AgentProfileKind, n as AgentProfileSource, o as AgentProfileSourceInput, J as JudgeScoresRecord, c as RunCostProvenance, e as RunJudgeMetadata, p as RunOutcome, q as RunRecordValidationError, d as RunTokenUsage, r as agentProfileCellHashMaterial, s as agentProfileCellKey, t as assertRunAgentProfileCell, u as buildAgentInterfaceProfileCell, v as buildAgentProfileCell, w as groupRunsByAgentProfileCell, x as isRunRecord, y as modelHasSnapshot, z as parseRunRecordSafe, B as requireAgentProfileCell, C as resolveRunCostProvenance, D as roundTripRunRecord, E as toAgentProfileJson, F as validateAgentProfileCell, G as validateRunRecord, H as verifyAgentProfileCell } from './run-record-B7RTi_ix.js';
|
|
4
|
+
import { B as BehavioralMetrics, A as RunScore, a as RunTrace, E as RunScoreWeights } from './semantic-concept-judge-DxJmRkyJ.js';
|
|
5
|
+
export { G as ConceptComplexity, H as ConceptFinding, J as ConceptSpec, L as ConceptWeightStrategy, C as CreateAnalystAiConfig, M as DEFAULT_COMPLEXITY_WEIGHTS, N as DEFAULT_RUN_SCORE_WEIGHTS, D as DEFAULT_TRACE_ANALYST_KINDS, c as DiffPolicy, F as FAILURE_MODE_KIND_SPEC, g as FindingSubject, h as FindingSubjectKind, j as FindingsDiff, k as FindingsStore, I as IMPROVEMENT_KIND_SPEC, l as KNOWLEDGE_GAP_KIND_SPEC, m as KNOWLEDGE_POISONING_KIND_SPEC, P as PersistedFinding, R as RunCritic, O as RunCriticOptions, Q as SEMANTIC_CONCEPT_JUDGE_VERSION, n as SKILL_USAGE_ANALYST, b as SemanticConceptJudgeInput, S as SemanticConceptJudgeOptions, T as SemanticConceptJudgeResult, o as SkillUsageAnalyst, U as SuboptimalCode, V as SuboptimalSignal, W as aggregateRunScore, X as clamp01, Y as computeTraceMetrics, t as createAnalystAi, Z as createSemanticConceptJudge, u as defaultIsMaterial, v as diffFindings, _ as runSemanticConceptJudge } from './semantic-concept-judge-DxJmRkyJ.js';
|
|
6
|
+
import { m as ChatRequest, q as CreateChatClientOpts } from './kind-factory-DcNg13sZ.js';
|
|
7
|
+
export { a as Analyst, b as AnalystContext, i as AnalystCost, A as AnalystFinding, j as AnalystInputKind, k as AnalystRequirements, f as AnalystRunEvent, e as AnalystRunInputs, d as AnalystRunResult, c as AnalystRunSummary, g as AnalystSeverity, l as ChatCallOpts, C as ChatClient, n as ChatResponse, o as ChatTransport, p as CliBridgeTransportOpts, r as CreateTraceAnalystKindOpts, D as DirectProviderTransportOpts, E as EvidenceRef, M as MockTransportOpts, s as RawAnalystFinding, u as RouterTransportOpts, S as SandboxSdkTransportOpts, v as TraceAnalystGolden, T as TraceAnalystKindSpec, w as computeFindingId, x as createChatClient, y as createTraceAnalystKind, z as makeFinding, F as renderPriorFindings } from './kind-factory-DcNg13sZ.js';
|
|
8
|
+
export { a as AnalystHooks, A as AnalystRegistry, b as AnalystRegistryOptions, B as BudgetPolicy, D as DefaultAnalystRegistryOptions, R as RegistryRunOpts, c as buildDefaultAnalystRegistry } from './default-registry-DltpYR5u.js';
|
|
9
|
+
export { F as FindingToPolicyEditOptions, P as POLICY_EDIT_AXES, a as POLICY_EDIT_TARGET_SURFACES, b as PolicyEdit, c as PolicyEditAdmission, d as PolicyEditAdmissionOptions, e as PolicyEditAxis, f as PolicyEditChange, g as PolicyEditExpectedGain, h as PolicyEditGainDirection, i as PolicyEditGainUnit, j as PolicyEditInit, k as PolicyEditRisk, l as PolicyEditSchemaVersion, m as PolicyEditSource, n as PolicyEditTarget, o as PolicyEditTargetSurface, p as PolicyEditValidationError, q as admitPolicyEdit, r as applyPolicyEditToSurface, s as computePolicyEditId, t as isPolicyEdit, u as makePolicyEdit, v as policyEditFromFinding, w as policyEditsFromFindings, x as scorePolicyEditReadiness, y as validatePolicyEdit } from './policy-edit-RLn8GWof.js';
|
|
10
10
|
import { TCloud } from '@tangle-network/tcloud';
|
|
11
11
|
import { B as BenchmarkRunnerConfig, S as Scenario, c as BenchmarkReport, P as ProductClientConfig, C as CheckResult, T as TestResult, d as PersonaConfig, D as DriverResult, e as DriverState, b as JudgeFn, f as CollectedArtifacts, g as ScenarioResult, h as TurnMetrics, i as ScenarioFile, j as CompletionCriterion } from './types-C7DGg5ex.js';
|
|
12
12
|
export { A as ArtifactCheck, k as ArtifactResult, E as EvalResult, F as FeedbackPattern, l as JudgeConfig, a as JudgeInput, m as JudgeRubric, J as JudgeScore, n as PersonaRigor, R as RouteMap, o as RubricDimension, p as Turn, q as TurnResult } from './types-C7DGg5ex.js';
|
|
@@ -16,12 +16,12 @@ import { F as FailureClass, T as ToolSpan, h as BudgetSpec, B as BudgetLedgerEnt
|
|
|
16
16
|
export { A as Artifact, E as EventKind, i as FAILURE_CLASSES, G as GenericSpan, J as JudgeSpan, M as Message, c as RetrievalSpan, g as RunLayer, f as RunStatus, d as SandboxSpan, j as SpanBase, b as SpanKind, k as SpanStatus, l as TRACE_SCHEMA_VERSION, e as TraceEvent, m as isJudgeSpan, n as isLlmSpan, o as isRetrievalSpan, p as isSandboxSpan, q as isToolSpan } from './schema-SGWcK9wa.js';
|
|
17
17
|
import { A as AgentEvalError, J as JudgeError, a as ConfigError } from './errors-oeQrLqXC.js';
|
|
18
18
|
export { b as AgentEvalErrorCode, C as CaptureIntegrityError, N as NotFoundError, R as ReplayError, V as ValidationError, c as VerificationError } from './errors-oeQrLqXC.js';
|
|
19
|
-
import { c as CorrectnessChecker } from './pre-registration-
|
|
20
|
-
export { A as ArtifactCheckArtifact, e as ArtifactEventLike, f as ArtifactValidator, g as BackendIntegrityError, B as BackendIntegrityReport, h as ComparePairedArmsOptions, C as CompletionRequirement, a as CompletionVerdict, H as HypothesisManifest, i as HypothesisResult, j as LlmCorrectnessCheckerOpts, L as LlmJudgeDimension, d as LlmJudgeOptions, M as MatchedPair, k as PairArmsOptions, m as PairArmsResult, n as PairedArmRow, P as PairedArmsComparison, o as PairedCorrectness, p as PairedMetricDelta, q as ProducedProposal, b as ProducedState, r as ProposalEventLike, s as RequirementCheck, R as RuntimeEventLike, t as SatisfiedBy, S as SignedManifest, u as SignedManifestAlgo, T as TaskGold, v as ToolCallEventLike, V as ValidationContext, w as ValidationIssue, x as ValidationResult, y as assertRealBackend, z as byteLengthRange, D as canonicalize, E as comparePairedArms, F as completionVerdict, G as composeValidators, I as containsAll, J as createLlmCorrectnessChecker, K as createTokenRecallChecker, N as evaluateHypothesis, O as extractProducedState, Q as hashJson, U as jsonHasKeys, l as llmJudge, W as pairArms, X as parseCorrectnessResponse, Y as regexMatch, Z as signManifest, _ as summarizeBackendIntegrity, $ as verifyCompletion, a0 as verifyManifest } from './pre-registration-
|
|
19
|
+
import { c as CorrectnessChecker } from './pre-registration-D8h7ZxNL.js';
|
|
20
|
+
export { A as ArtifactCheckArtifact, e as ArtifactEventLike, f as ArtifactValidator, g as BackendIntegrityError, B as BackendIntegrityReport, h as ComparePairedArmsOptions, C as CompletionRequirement, a as CompletionVerdict, H as HypothesisManifest, i as HypothesisResult, j as LlmCorrectnessCheckerOpts, L as LlmJudgeDimension, d as LlmJudgeOptions, M as MatchedPair, k as PairArmsOptions, m as PairArmsResult, n as PairedArmRow, P as PairedArmsComparison, o as PairedCorrectness, p as PairedMetricDelta, q as ProducedProposal, b as ProducedState, r as ProposalEventLike, s as RequirementCheck, R as RuntimeEventLike, t as SatisfiedBy, S as SignedManifest, u as SignedManifestAlgo, T as TaskGold, v as ToolCallEventLike, V as ValidationContext, w as ValidationIssue, x as ValidationResult, y as assertRealBackend, z as byteLengthRange, D as canonicalize, E as comparePairedArms, F as completionVerdict, G as composeValidators, I as containsAll, J as createLlmCorrectnessChecker, K as createTokenRecallChecker, N as evaluateHypothesis, O as extractProducedState, Q as hashJson, U as jsonHasKeys, l as llmJudge, W as pairArms, X as parseCorrectnessResponse, Y as regexMatch, Z as signManifest, _ as summarizeBackendIntegrity, $ as verifyCompletion, a0 as verifyManifest } from './pre-registration-D8h7ZxNL.js';
|
|
21
21
|
import { T as TraceEmitter } from './emitter-BRchAAAx.js';
|
|
22
22
|
export { R as RunCompleteHook, a as RunCompleteHookContext, S as SpanHandle, b as TraceEmitterOptions, l as llmSpanFromProvider } from './emitter-BRchAAAx.js';
|
|
23
|
-
import { h as ReleaseConfidenceThresholds, f as ReleaseConfidenceScorecard } from './release-report-
|
|
24
|
-
export { A as ActionableSideInfo, o as AsiSeverity, B as BootstrapOptions, a as BootstrapResult, J as JudgeReplayGateArgs, R as ReleaseConfidenceAxis, b as ReleaseConfidenceAxisName, c as ReleaseConfidenceInput, d as ReleaseConfidenceIssue, e as ReleaseConfidenceMetrics, g as ReleaseConfidenceStatus, i as ReleaseTraceEvidence, j as RenderReleaseReportOptions, V as Verdict, k as assertReleaseConfidence, l as bootstrapCi, m as evaluateReleaseConfidence, n as judgeReplayGate, r as renderReleaseReport } from './release-report-
|
|
23
|
+
import { h as ReleaseConfidenceThresholds, f as ReleaseConfidenceScorecard } from './release-report-CCtzajxP.js';
|
|
24
|
+
export { A as ActionableSideInfo, o as AsiSeverity, B as BootstrapOptions, a as BootstrapResult, J as JudgeReplayGateArgs, R as ReleaseConfidenceAxis, b as ReleaseConfidenceAxisName, c as ReleaseConfidenceInput, d as ReleaseConfidenceIssue, e as ReleaseConfidenceMetrics, g as ReleaseConfidenceStatus, i as ReleaseTraceEvidence, j as RenderReleaseReportOptions, V as Verdict, k as assertReleaseConfidence, l as bootstrapCi, m as evaluateReleaseConfidence, n as judgeReplayGate, r as renderReleaseReport } from './release-report-CCtzajxP.js';
|
|
25
25
|
export { c as CliffsMagnitude, d as CorpusAgreementOptions, e as CorpusAgreementPerDimension, C as CorpusAgreementReport, f as CorpusScoreRecord, g as EProcess, h as EProcessOptions, E as EProcessState, i as EProcessStep, M as McNemarResult, P as PairedBootstrapOptions, a as PairedBootstrapResult, j as PairedSignTestResult, k as ProportionInterval, R as RiskDifferenceResult, S as SignTestAlternative, W as WeightedCompositeInput, l as WeightedCompositeResult, b as benjaminiHochberg, m as bonferroni, n as cliffsDelta, o as cohensD, q as confidenceInterval, r as corpusInterRaterAgreement, s as corpusInterRaterAgreementFromJudgeScores, t as eProcess, u as holm, v as interRaterReliability, x as interpretCliffs, y as mannWhitneyU, z as mcnemar, A as mcnemarPower, B as mcnemarRequiredN, D as mulberry32, F as normalizeScores, p as pairedBootstrap, G as pairedMde, H as pairedRiskDifference, I as pairedSignTest, J as pairedTTest, K as partialCredit, L as passAtK, N as pearsonR, O as ranks, Q as requiredSampleSize, T as spearmanR, U as weightedComposite, V as weightedMean, w as wilcoxonSignedRank, X as wilson } from './statistics-oUbOJe-S.js';
|
|
26
26
|
import { OtelExporter, OtelExportConfig } from './traces.js';
|
|
27
27
|
export { CaptureFetchContext, CaptureFetchOptions, DEFAULT_REDACTION_RULES, ExportableSpan, ExtractedUsage, FlattenOtlpOptions, LLM_CACHED_TOKENS, LLM_CACHED_TOKEN_ATTR_KEYS, LLM_COST_ATTR_KEYS, LLM_COST_USD, LLM_INPUT_TOKENS, LLM_INPUT_TOKEN_ATTR_KEYS, LLM_MODEL_ATTR_KEYS, LLM_MODEL_NAME, LLM_OUTPUT_TOKENS, LLM_OUTPUT_TOKEN_ATTR_KEYS, OPENINFERENCE_SPAN_KIND, OTEL_AGENT_EVAL_SCOPE, OtlpExport, OtlpFileTraceStore, OtlpFileTraceStoreOptions, OtlpFlatLine, OtlpResourceSpans, OtlpSpan, OtlpToRunRecordsOptions, OtlpTraceRunRecord, ProjectedOtlpSpan, REDACTION_VERSION, RedactionReport, RedactionRule, ReplayCache, ReplayCacheEntry, ReplayCacheMissError, ReplayCacheStats, ReplayFetchOptions, SPAN_KIND_ATTR_KEYS, SpanNotFoundError, TOOL_NAME, TOOL_NAME_ATTR_KEYS, TRACE_ANALYST_ACTOR_DESCRIPTION, TRACE_ANALYST_ACTOR_DESCRIPTION_VERSION, TRACE_ANALYST_SUBAGENT_DESCRIPTION, TraceAggregate, TraceAnalystHookOptions, TraceFileMissingError, TraceInsightContext, TraceInsightFinding, TraceInsightPanelRole, TraceInsightPromptInput, TraceInsightQualityGate, TraceInsightQuestion, TraceInsightReadiness, TraceInsightSuite, TraceInsightTask, TraceNotFoundError, TraceStoreSource, TraceStoreToOtlpOptions, TracesToOtlpResult, asNumber, asString, buildTraceAnalystTools, buildTraceInsightContext, buildTraceInsightPrompt, captureFetchToRawSink, convertTraceStoresToOtlp, createOtelExporter, createOtelTracingStore, createReplayFetch, defaultTraceInsightPanel, describeTraceInsightScope, domainEvidencePattern, exportRunAsOtlp, extractOtlpAttributes, extractUsage, extractUsageFromResponse, extractUsageFromSse, firstNumberAttr, firstStringAttr, flattenOtlpExportToNdjson, inferDomainKeywords, inferOtlpKind, iterateRawCalls, otelRunCompleteHook, otlpToRunRecords, otlpToTraceRunRecords, planTraceInsightQuestions, projectOtlpFlatLine, readOtlpStatus, redactString, redactValue, scoreTraceInsightReadiness, stringField, tokenizeDomainWords, traceAnalystFunctionGroup, traceAnalystOnRunComplete, traceSpanKindToOpenInferenceKind } from './traces.js';
|
|
@@ -29,10 +29,10 @@ import { a as AnalyzeTracesInput, A as AnalyzeTracesOptions, b as AnalyzeTracesR
|
|
|
29
29
|
export { c as AnalyzeTracesTurnSnapshot, d as analyzeTraces } from './analyst-C8HHvfJp.js';
|
|
30
30
|
import { a as TraceAnalystSpan } from './store-C1YxJDEK.js';
|
|
31
31
|
export { D as DEFAULT_TRACE_ANALYST_BUDGETS, b as DatasetOverview, E as ErrorCluster, Q as QueryTracesPage, S as SearchSpanResult, c as SearchTraceResult, d as SpanMatchRecord, e as TRACE_ANALYST_TRUNCATION_MARKER_PREFIX, T as TraceAnalysisStore, f as TraceAnalystByteBudgets, g as TraceAnalystFilters, h as TraceAnalystSpanKind, i as TraceAnalystSpanStatus, j as TraceAnalystTraceSummary, V as ViewSpansResult, k as ViewTraceOversized, l as ViewTraceResult } from './store-C1YxJDEK.js';
|
|
32
|
-
import { b as JudgeConfig, J as JudgeScore, S as Scenario$1, g as Gate } from './types-
|
|
33
|
-
import { A as AnalyzeRunsOptions } from './analyze-runs-
|
|
34
|
-
import { q as Objective, s as ParetoResult, h as GepaProposerConstraints, a as RunImprovementLoopResult } from './gepa-
|
|
35
|
-
export { t as DEFAULT_RED_TEAM_CORPUS, D as Direction, e as RedTeamCase, u as RedTeamCategory, v as RedTeamFinding, w as RedTeamPayload, x as RedTeamReport, y as crowdingDistance, z as dominates, A as paretoFrontier, B as paretoFrontierWithCrowding, E as redTeamDataset, F as redTeamReport, H as scalarScore, I as scoreRedTeamOutput, J as toolNamesForRun } from './gepa-
|
|
32
|
+
import { b as JudgeConfig, J as JudgeScore, S as Scenario$1, g as Gate } from './types-C5gJrOVT.js';
|
|
33
|
+
import { A as AnalyzeRunsOptions } from './analyze-runs-BYHg6Irm.js';
|
|
34
|
+
import { q as Objective, s as ParetoResult, h as GepaProposerConstraints, a as RunImprovementLoopResult } from './gepa-dne9JDPL.js';
|
|
35
|
+
export { t as DEFAULT_RED_TEAM_CORPUS, D as Direction, e as RedTeamCase, u as RedTeamCategory, v as RedTeamFinding, w as RedTeamPayload, x as RedTeamReport, y as crowdingDistance, z as dominates, A as paretoFrontier, B as paretoFrontierWithCrowding, E as redTeamDataset, F as redTeamReport, H as scalarScore, I as scoreRedTeamOutput, J as toolNamesForRun } from './gepa-dne9JDPL.js';
|
|
36
36
|
import { S as SandboxDriver, H as HarnessConfig, a as SandboxHarnessResult } from './test-graded-scenario-mzYBKspu.js';
|
|
37
37
|
export { D as DockerSandboxDriver, c as SandboxHarness, d as SandboxResult, e as SubprocessSandboxDriver, f as SubprocessSandboxDriverOptions, g as TestGradedRunOptions, b as TestGradedRunResult, T as TestGradedScenario, h as TestOutputParser, i as composeParsers, j as jestTestParser, p as pytestTestParser, r as runTestGradedScenario, v as vitestTestParser } from './test-graded-scenario-mzYBKspu.js';
|
|
38
38
|
export { b as RunIntegrityError, R as RunIntegrityExpectations, c as RunIntegrityIssue, d as RunIntegrityIssueCode, a as RunIntegrityReport, e as assertRunCaptured, t as throwIfRunIncomplete } from './integrity-qemeBAyx.js';
|
|
@@ -41,7 +41,7 @@ export { F as FileSystemRawProviderSink, a as FileSystemRawProviderSinkOptions,
|
|
|
41
41
|
import { T as TraceStore, R as RunFilter } from './store-BsVi7ncX.js';
|
|
42
42
|
export { E as EventFilter, F as FileSystemTraceStore, a as FileSystemTraceStoreOptions, I as InMemoryTraceStore, S as SpanFilter } from './store-BsVi7ncX.js';
|
|
43
43
|
export { D as DEFAULT_FAILURE_RULES, b as FailureClassification, c as FailureContext, d as FailureRule, e as classifyFailure } from './failure-cluster-C48PiReX.js';
|
|
44
|
-
export { P as ProjectRuntimeTrajectoryEvidenceOptions, a as RuntimeTrajectoryEvidenceProjection, b as RuntimeTrajectoryEvidenceSummary, c as RuntimeTrajectoryHookEvent, R as RuntimeTrajectoryRecord, d as RuntimeTrajectoryRunRecord, p as parseRuntimeTrajectoryHookEvent, e as projectRuntimeTrajectoryEvidence } from './runtime-trajectory-
|
|
44
|
+
export { P as ProjectRuntimeTrajectoryEvidenceOptions, a as RuntimeTrajectoryEvidenceProjection, b as RuntimeTrajectoryEvidenceSummary, c as RuntimeTrajectoryHookEvent, R as RuntimeTrajectoryRecord, d as RuntimeTrajectoryRunRecord, p as parseRuntimeTrajectoryHookEvent, e as projectRuntimeTrajectoryEvidence } from './runtime-trajectory-Dws7Kpgi.js';
|
|
45
45
|
import { a as BaselineReport, b as Trajectory, T as TrajectoryStep } from './baseline-DsNteOgR.js';
|
|
46
46
|
export { B as BaselineOptions, M as MetricSamples, d as MetricVerdict, e as ToolStats, f as ToolUseMetrics, g as ToolUseOptions, h as buildTrajectory, i as compareToBaseline, c as computeToolUseMetrics, j as iqr, w as welchsTTest } from './baseline-DsNteOgR.js';
|
|
47
47
|
import { HarnessType, AgentProfile } from '@tangle-network/agent-interface';
|
|
@@ -57,13 +57,13 @@ import { L as Layer, S as Severity, b as LayerResult, c as VerifyContext } from
|
|
|
57
57
|
export { F as Finding, d as LayerStatus, M as MultiLayerVerifier, a as VerificationReport, V as VerifyOptions, g as gradeSemanticStatus } from './multi-layer-verifier-BsqKuLyN.js';
|
|
58
58
|
import { L as LlmClientOptions } from './llm-client-DyqEH4jH.js';
|
|
59
59
|
export { d as LlmCallError, b as LlmCallRequest, c as LlmCallResult, e as LlmClient, f as LlmMessage, g as LlmRouteAssertionError, a as LlmRouteRequirements, h as LlmUsage, i as assertLlmRoute, j as backoffMs, k as callLlm, l as callLlmJson, m as isTransientLlmError, p as probeLlm, s as stripFencedJson } from './llm-client-DyqEH4jH.js';
|
|
60
|
-
export { B as BENCHMARK_SPLIT_SEED, a as BenchmarkAdapter, b as BenchmarkDatasetItem, c as BenchmarkEvaluation, d as BenchmarkFamily, e as BenchmarkResponder, f as BenchmarkScenario, g as BenchmarkSource, h as BenchmarkTaskKind, i as benchmarkDeterministicSplit, j as benchmarks } from './index-
|
|
61
|
-
export { C as CallbackResearcher, d as CallbackResearcherOptions, e as CampaignFactoryParams, f as CampaignIntegrityPolicy, g as CampaignRunContext, h as CampaignRunOutcome, i as CampaignRunner, j as CampaignScenario, k as CampaignVariant, c as EvalCampaignOptions, b as EvalCampaignResult, E as ExperimentPlan, a as ExperimentResult, l as FailedRun, F as FailureMode, N as NoopResearcher, R as Researcher, S as SteeringChange, r as runEvalCampaign } from './researcher-
|
|
62
|
-
export { G as GainDistributionBin, a as GainDistributionFigureSpec, b as GainDistributionOptions, m as GateDecision, n as GateEvidence, H as HeldOutGate, o as HeldOutGateConfig, q as HeldOutGateRejectionCode, P as ParetoFigureSpec, c as ParetoPoint, R as RESEARCH_REPORT_HARD_PAIR_FLOOR, d as ResearchReport, e as ResearchReportCandidate, f as ResearchReportDecision, g as ResearchReportMethodology, h as ResearchReportOptions, i as ResearchReportRecommendation, S as SummaryTable, j as SummaryTableOptions, k as SummaryTableRow, l as gainHistogram, p as paretoChart, r as researchReport, s as summaryTable } from './summary-report-
|
|
60
|
+
export { B as BENCHMARK_SPLIT_SEED, a as BenchmarkAdapter, b as BenchmarkDatasetItem, c as BenchmarkEvaluation, d as BenchmarkFamily, e as BenchmarkResponder, f as BenchmarkScenario, g as BenchmarkSource, h as BenchmarkTaskKind, i as benchmarkDeterministicSplit, j as benchmarks } from './index-BTEpx9He.js';
|
|
61
|
+
export { C as CallbackResearcher, d as CallbackResearcherOptions, e as CampaignFactoryParams, f as CampaignIntegrityPolicy, g as CampaignRunContext, h as CampaignRunOutcome, i as CampaignRunner, j as CampaignScenario, k as CampaignVariant, c as EvalCampaignOptions, b as EvalCampaignResult, E as ExperimentPlan, a as ExperimentResult, l as FailedRun, F as FailureMode, N as NoopResearcher, R as Researcher, S as SteeringChange, r as runEvalCampaign } from './researcher-Dq-EtpbE.js';
|
|
62
|
+
export { G as GainDistributionBin, a as GainDistributionFigureSpec, b as GainDistributionOptions, m as GateDecision, n as GateEvidence, H as HeldOutGate, o as HeldOutGateConfig, q as HeldOutGateRejectionCode, P as ParetoFigureSpec, c as ParetoPoint, R as RESEARCH_REPORT_HARD_PAIR_FLOOR, d as ResearchReport, e as ResearchReportCandidate, f as ResearchReportDecision, g as ResearchReportMethodology, h as ResearchReportOptions, i as ResearchReportRecommendation, S as SummaryTable, j as SummaryTableOptions, k as SummaryTableRow, l as gainHistogram, p as paretoChart, r as researchReport, s as summaryTable } from './summary-report-BJ5aNwZ1.js';
|
|
63
63
|
export { I as InterimReleaseConfidence, a as InterimReleaseConfidenceInput, P as PairedEvalueOptions, b as PairedEvalueSequence, c as PairedEvalueStep, S as SequentialDecision, e as evaluateInterimReleaseConfidence, p as pairedEvalueSequence } from './sequential-5iSVfzl2.js';
|
|
64
64
|
import '@ax-llm/ax';
|
|
65
65
|
import 'zod';
|
|
66
|
-
import './insight-report-
|
|
66
|
+
import './insight-report-IwwvqZZv.js';
|
|
67
67
|
import './storage-Dw_f7WMt.js';
|
|
68
68
|
|
|
69
69
|
/**
|
package/dist/index.js
CHANGED
|
@@ -9,7 +9,7 @@ import {
|
|
|
9
9
|
checkBehavioralCanary,
|
|
10
10
|
checkCanaries,
|
|
11
11
|
runBehavioralCanaries
|
|
12
|
-
} from "./chunk-
|
|
12
|
+
} from "./chunk-J6P6PK2R.js";
|
|
13
13
|
import {
|
|
14
14
|
BENCHMARK_SPLIT_SEED,
|
|
15
15
|
benchmarks_exports,
|
|
@@ -61,7 +61,7 @@ import {
|
|
|
61
61
|
pairArms,
|
|
62
62
|
parseCorrectnessResponse,
|
|
63
63
|
verifyCompletion
|
|
64
|
-
} from "./chunk-
|
|
64
|
+
} from "./chunk-5NVBGKPH.js";
|
|
65
65
|
import {
|
|
66
66
|
DEFAULT_MUTATION_PRIMITIVES,
|
|
67
67
|
DEFAULT_RED_TEAM_CORPUS,
|
|
@@ -174,7 +174,7 @@ import {
|
|
|
174
174
|
stopOnNoProgress,
|
|
175
175
|
stopOnRepeatedAction,
|
|
176
176
|
subjectiveEval
|
|
177
|
-
} from "./chunk-
|
|
177
|
+
} from "./chunk-DXZRATT5.js";
|
|
178
178
|
import {
|
|
179
179
|
assertReleaseConfidence,
|
|
180
180
|
bootstrapCi,
|
|
@@ -184,7 +184,7 @@ import {
|
|
|
184
184
|
} from "./chunk-MOXWMGPC.js";
|
|
185
185
|
import {
|
|
186
186
|
runEvalCampaign
|
|
187
|
-
} from "./chunk-
|
|
187
|
+
} from "./chunk-QG5F6463.js";
|
|
188
188
|
import "./chunk-ARU2PZFM.js";
|
|
189
189
|
import {
|
|
190
190
|
evaluateInterimReleaseConfidence,
|
|
@@ -267,7 +267,7 @@ import {
|
|
|
267
267
|
scoreTraceInsightReadiness,
|
|
268
268
|
tokenizeDomainWords,
|
|
269
269
|
traceAnalystOnRunComplete
|
|
270
|
-
} from "./chunk-
|
|
270
|
+
} from "./chunk-TLDB7WRY.js";
|
|
271
271
|
import {
|
|
272
272
|
FAILURE_CLASSES,
|
|
273
273
|
TRACE_SCHEMA_VERSION,
|
|
@@ -345,9 +345,10 @@ import {
|
|
|
345
345
|
isRunRecord,
|
|
346
346
|
modelHasSnapshot,
|
|
347
347
|
parseRunRecordSafe,
|
|
348
|
+
resolveRunCostProvenance,
|
|
348
349
|
roundTripRunRecord,
|
|
349
350
|
validateRunRecord
|
|
350
|
-
} from "./chunk-
|
|
351
|
+
} from "./chunk-5UF54T55.js";
|
|
351
352
|
import {
|
|
352
353
|
AGENT_PROFILE_KINDS,
|
|
353
354
|
AgentProfileCellValidationError,
|
|
@@ -11924,6 +11925,7 @@ export {
|
|
|
11924
11925
|
requiredSampleSize,
|
|
11925
11926
|
researchReport,
|
|
11926
11927
|
resolveModelPricing,
|
|
11928
|
+
resolveRunCostProvenance,
|
|
11927
11929
|
resolveSeat,
|
|
11928
11930
|
roundTripRunRecord,
|
|
11929
11931
|
routeFields,
|