@tangle-network/agent-eval 0.115.1 → 0.115.3

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (79) hide show
  1. package/CHANGELOG.md +20 -0
  2. package/dist/analyst/index.d.ts +7 -7
  3. package/dist/analyst/index.js +5 -5
  4. package/dist/{analyze-runs-Dmz6LA9e.d.ts → analyze-runs-BYHg6Irm.d.ts} +3 -3
  5. package/dist/belief-state/index.d.ts +3 -3
  6. package/dist/belief-state/index.js +1 -1
  7. package/dist/benchmarks/index.d.ts +3 -3
  8. package/dist/benchmarks/index.js +6 -6
  9. package/dist/campaign/index.d.ts +12 -12
  10. package/dist/campaign/index.js +6 -6
  11. package/dist/{chunk-DRPIZQIT.js → chunk-4D5RVB3W.js} +2 -2
  12. package/dist/{chunk-DWLIGZBX.js → chunk-5S5NJ63F.js} +2 -2
  13. package/dist/{chunk-VK6HBGAE.js → chunk-5UF54T55.js} +53 -1
  14. package/dist/chunk-5UF54T55.js.map +1 -0
  15. package/dist/{chunk-N6MTC3GK.js → chunk-ADYLPOSX.js} +2 -2
  16. package/dist/{chunk-IMWDSFUM.js → chunk-DXZRATT5.js} +2 -2
  17. package/dist/{chunk-S42AWHMP.js → chunk-E4BUPP7Z.js} +135 -17
  18. package/dist/chunk-E4BUPP7Z.js.map +1 -0
  19. package/dist/{chunk-FUCQVFMU.js → chunk-GY4SYVPJ.js} +12 -3
  20. package/dist/chunk-GY4SYVPJ.js.map +1 -0
  21. package/dist/{chunk-LVTGFSHF.js → chunk-I6LVHOV3.js} +2 -2
  22. package/dist/{chunk-WBOGKYM4.js → chunk-J6P6PK2R.js} +48 -7
  23. package/dist/chunk-J6P6PK2R.js.map +1 -0
  24. package/dist/{chunk-KDCMEZDI.js → chunk-KG4TD7EQ.js} +6 -6
  25. package/dist/{chunk-LOBMT6SB.js → chunk-ONM6PEAE.js} +3 -3
  26. package/dist/{chunk-AN5UYSVD.js → chunk-QMXXSNC4.js} +2 -2
  27. package/dist/{chunk-RSVSSZKF.js → chunk-TLDB7WRY.js} +2 -2
  28. package/dist/{chunk-I2HNIE6N.js → chunk-WSBUZMBU.js} +4 -4
  29. package/dist/cli.js +2 -2
  30. package/dist/{code-agent-session-yitf9I-F.d.ts → code-agent-session-D-g04tcy.d.ts} +8 -1
  31. package/dist/contract/index.d.ts +15 -15
  32. package/dist/contract/index.js +7 -7
  33. package/dist/{control-U8LBKUES.d.ts → control-CcBiAEnn.d.ts} +1 -1
  34. package/dist/control.d.ts +2 -2
  35. package/dist/control.js +2 -2
  36. package/dist/{default-registry-Bcf1uKVI.d.ts → default-registry-DltpYR5u.d.ts} +1 -1
  37. package/dist/{gepa-DolL_Fko.d.ts → gepa-dne9JDPL.d.ts} +1 -1
  38. package/dist/hosted/index.d.ts +4 -4
  39. package/dist/{index-CWr5SIG-.d.ts → index-BTEpx9He.d.ts} +2 -2
  40. package/dist/index.d.ts +22 -22
  41. package/dist/index.js +14 -12
  42. package/dist/index.js.map +1 -1
  43. package/dist/{insight-report-D4cXFsLt.d.ts → insight-report-IwwvqZZv.d.ts} +20 -2
  44. package/dist/{kind-factory-20hcaYpf.d.ts → kind-factory-DcNg13sZ.d.ts} +1 -1
  45. package/dist/meta-eval/index.d.ts +2 -2
  46. package/dist/multishot/index.d.ts +2 -2
  47. package/dist/openapi.json +1 -1
  48. package/dist/{policy-edit-az2qRmvN.d.ts → policy-edit-RLn8GWof.d.ts} +2 -2
  49. package/dist/{pre-registration-oNItiRBb.d.ts → pre-registration-D8h7ZxNL.d.ts} +3 -3
  50. package/dist/{provenance-BZmpWmn4.d.ts → provenance-Bibyg1U9.d.ts} +3 -3
  51. package/dist/{release-report-oBfOz8ku.d.ts → release-report-CCtzajxP.d.ts} +2 -2
  52. package/dist/reporting.d.ts +4 -4
  53. package/dist/{researcher-CaH0CwFC.d.ts → researcher-Dq-EtpbE.d.ts} +2 -2
  54. package/dist/rl.d.ts +6 -6
  55. package/dist/rl.js +3 -3
  56. package/dist/{rubric-predictive-validity-C-fMteAW.d.ts → rubric-predictive-validity-DYTLjGWu.d.ts} +1 -1
  57. package/dist/{run-record-DksGsfgv.d.ts → run-record-B7RTi_ix.d.ts} +34 -2
  58. package/dist/{runtime-trajectory-h5i0SZUj.d.ts → runtime-trajectory-Dws7Kpgi.d.ts} +1 -1
  59. package/dist/{semantic-concept-judge-CpzbtwD0.d.ts → semantic-concept-judge-DxJmRkyJ.d.ts} +1 -1
  60. package/dist/{summary-report-Bz-0-t8v.d.ts → summary-report-BJ5aNwZ1.d.ts} +1 -1
  61. package/dist/traces.d.ts +1 -1
  62. package/dist/traces.js +2 -2
  63. package/dist/{types-CgSlO6wT.d.ts → types-C5gJrOVT.d.ts} +1 -1
  64. package/dist/wire/index.js +2 -2
  65. package/package.json +1 -1
  66. package/dist/chunk-FUCQVFMU.js.map +0 -1
  67. package/dist/chunk-S42AWHMP.js.map +0 -1
  68. package/dist/chunk-VK6HBGAE.js.map +0 -1
  69. package/dist/chunk-WBOGKYM4.js.map +0 -1
  70. /package/dist/{chunk-DRPIZQIT.js.map → chunk-4D5RVB3W.js.map} +0 -0
  71. /package/dist/{chunk-DWLIGZBX.js.map → chunk-5S5NJ63F.js.map} +0 -0
  72. /package/dist/{chunk-N6MTC3GK.js.map → chunk-ADYLPOSX.js.map} +0 -0
  73. /package/dist/{chunk-IMWDSFUM.js.map → chunk-DXZRATT5.js.map} +0 -0
  74. /package/dist/{chunk-LVTGFSHF.js.map → chunk-I6LVHOV3.js.map} +0 -0
  75. /package/dist/{chunk-KDCMEZDI.js.map → chunk-KG4TD7EQ.js.map} +0 -0
  76. /package/dist/{chunk-LOBMT6SB.js.map → chunk-ONM6PEAE.js.map} +0 -0
  77. /package/dist/{chunk-AN5UYSVD.js.map → chunk-QMXXSNC4.js.map} +0 -0
  78. /package/dist/{chunk-RSVSSZKF.js.map → chunk-TLDB7WRY.js.map} +0 -0
  79. /package/dist/{chunk-I2HNIE6N.js.map → chunk-WSBUZMBU.js.map} +0 -0
@@ -1,10 +1,10 @@
1
- import { M as MutableSurface, h as GateDecision } from '../types-CgSlO6wT.js';
2
- import { I as InsightReport } from '../insight-report-D4cXFsLt.js';
3
- import '../run-record-DksGsfgv.js';
1
+ import { M as MutableSurface, h as GateDecision } from '../types-C5gJrOVT.js';
2
+ import { I as InsightReport } from '../insight-report-IwwvqZZv.js';
3
+ import '../run-record-B7RTi_ix.js';
4
4
  import '@tangle-network/agent-interface';
5
5
  import '../errors-oeQrLqXC.js';
6
6
  import '../schema-SGWcK9wa.js';
7
- import '../summary-report-Bz-0-t8v.js';
7
+ import '../summary-report-BJ5aNwZ1.js';
8
8
  import '../failure-cluster-C48PiReX.js';
9
9
  import '../store-BsVi7ncX.js';
10
10
  import '../judge-calibration-7C-IDmKr.js';
@@ -1,5 +1,5 @@
1
- import { S as Scenario, D as DispatchContext, C as CampaignResult } from './types-CgSlO6wT.js';
2
- import { b as RunSplitTag } from './run-record-DksGsfgv.js';
1
+ import { S as Scenario, D as DispatchContext, C as CampaignResult } from './types-C5gJrOVT.js';
2
+ import { b as RunSplitTag } from './run-record-B7RTi_ix.js';
3
3
  import { C as CampaignStorage } from './storage-Dw_f7WMt.js';
4
4
 
5
5
  /**
package/dist/index.d.ts CHANGED
@@ -1,12 +1,12 @@
1
- export { A as ActionExecutionPolicy, a as ActionPolicyDecision, C as ControlRunToRunRecordOptions, L as LlmJsonCall, b as LlmReviewerConfig, P as ProposeFn, c as ProposeInput, d as ProposeOutput, e as ProposeReviewConfig, f as ProposeReviewControlAction, g as ProposeReviewControlConfig, h as ProposeReviewControlResult, i as ProposeReviewControlState, j as ProposeReviewReport, k as ProposeReviewShot, R as Review, l as ReviewFn, m as ReviewInput, n as ReviewMemoryEntry, o as ReviewMemoryStore, p as RunEvidenceMetadata, V as Verification, q as VerifyFn, r as controlFailureClassFromVerification, s as controlRunToRunRecord, t as createLlmReviewer, u as evaluateActionPolicy, v as inMemoryReviewStore, w as jsonlReviewStore, x as runProposeReview, y as runProposeReviewAsControlLoop, z as scoreFromEvals } from './control-U8LBKUES.js';
2
- import { R as RunRecord, b as RunSplitTag } from './run-record-DksGsfgv.js';
3
- export { f as AGENT_PROFILE_KINDS, g as AgentInterfaceProfileLike, A as AgentProfileCell, e as AgentProfileCellInput, h as AgentProfileCellSchemaVersion, i as AgentProfileCellValidationError, j as AgentProfileDimensionValue, k as AgentProfileHarness, a as AgentProfileJson, l as AgentProfileKind, m as AgentProfileSource, n as AgentProfileSourceInput, J as JudgeScoresRecord, d as RunJudgeMetadata, o as RunOutcome, p as RunRecordValidationError, c as RunTokenUsage, q as agentProfileCellHashMaterial, r as agentProfileCellKey, s as assertRunAgentProfileCell, t as buildAgentInterfaceProfileCell, u as buildAgentProfileCell, v as groupRunsByAgentProfileCell, w as isRunRecord, x as modelHasSnapshot, y as parseRunRecordSafe, z as requireAgentProfileCell, B as roundTripRunRecord, C as toAgentProfileJson, D as validateAgentProfileCell, E as validateRunRecord, F as verifyAgentProfileCell } from './run-record-DksGsfgv.js';
4
- import { B as BehavioralMetrics, A as RunScore, a as RunTrace, E as RunScoreWeights } from './semantic-concept-judge-CpzbtwD0.js';
5
- export { G as ConceptComplexity, H as ConceptFinding, J as ConceptSpec, L as ConceptWeightStrategy, C as CreateAnalystAiConfig, M as DEFAULT_COMPLEXITY_WEIGHTS, N as DEFAULT_RUN_SCORE_WEIGHTS, D as DEFAULT_TRACE_ANALYST_KINDS, c as DiffPolicy, F as FAILURE_MODE_KIND_SPEC, g as FindingSubject, h as FindingSubjectKind, j as FindingsDiff, k as FindingsStore, I as IMPROVEMENT_KIND_SPEC, l as KNOWLEDGE_GAP_KIND_SPEC, m as KNOWLEDGE_POISONING_KIND_SPEC, P as PersistedFinding, R as RunCritic, O as RunCriticOptions, Q as SEMANTIC_CONCEPT_JUDGE_VERSION, n as SKILL_USAGE_ANALYST, b as SemanticConceptJudgeInput, S as SemanticConceptJudgeOptions, T as SemanticConceptJudgeResult, o as SkillUsageAnalyst, U as SuboptimalCode, V as SuboptimalSignal, W as aggregateRunScore, X as clamp01, Y as computeTraceMetrics, t as createAnalystAi, Z as createSemanticConceptJudge, u as defaultIsMaterial, v as diffFindings, _ as runSemanticConceptJudge } from './semantic-concept-judge-CpzbtwD0.js';
6
- import { m as ChatRequest, q as CreateChatClientOpts } from './kind-factory-20hcaYpf.js';
7
- export { a as Analyst, b as AnalystContext, i as AnalystCost, A as AnalystFinding, j as AnalystInputKind, k as AnalystRequirements, f as AnalystRunEvent, e as AnalystRunInputs, d as AnalystRunResult, c as AnalystRunSummary, g as AnalystSeverity, l as ChatCallOpts, C as ChatClient, n as ChatResponse, o as ChatTransport, p as CliBridgeTransportOpts, r as CreateTraceAnalystKindOpts, D as DirectProviderTransportOpts, E as EvidenceRef, M as MockTransportOpts, s as RawAnalystFinding, u as RouterTransportOpts, S as SandboxSdkTransportOpts, v as TraceAnalystGolden, T as TraceAnalystKindSpec, w as computeFindingId, x as createChatClient, y as createTraceAnalystKind, z as makeFinding, F as renderPriorFindings } from './kind-factory-20hcaYpf.js';
8
- export { a as AnalystHooks, A as AnalystRegistry, b as AnalystRegistryOptions, B as BudgetPolicy, D as DefaultAnalystRegistryOptions, R as RegistryRunOpts, c as buildDefaultAnalystRegistry } from './default-registry-Bcf1uKVI.js';
9
- export { F as FindingToPolicyEditOptions, P as POLICY_EDIT_AXES, a as POLICY_EDIT_TARGET_SURFACES, b as PolicyEdit, c as PolicyEditAdmission, d as PolicyEditAdmissionOptions, e as PolicyEditAxis, f as PolicyEditChange, g as PolicyEditExpectedGain, h as PolicyEditGainDirection, i as PolicyEditGainUnit, j as PolicyEditInit, k as PolicyEditRisk, l as PolicyEditSchemaVersion, m as PolicyEditSource, n as PolicyEditTarget, o as PolicyEditTargetSurface, p as PolicyEditValidationError, q as admitPolicyEdit, r as applyPolicyEditToSurface, s as computePolicyEditId, t as isPolicyEdit, u as makePolicyEdit, v as policyEditFromFinding, w as policyEditsFromFindings, x as scorePolicyEditReadiness, y as validatePolicyEdit } from './policy-edit-az2qRmvN.js';
1
+ export { A as ActionExecutionPolicy, a as ActionPolicyDecision, C as ControlRunToRunRecordOptions, L as LlmJsonCall, b as LlmReviewerConfig, P as ProposeFn, c as ProposeInput, d as ProposeOutput, e as ProposeReviewConfig, f as ProposeReviewControlAction, g as ProposeReviewControlConfig, h as ProposeReviewControlResult, i as ProposeReviewControlState, j as ProposeReviewReport, k as ProposeReviewShot, R as Review, l as ReviewFn, m as ReviewInput, n as ReviewMemoryEntry, o as ReviewMemoryStore, p as RunEvidenceMetadata, V as Verification, q as VerifyFn, r as controlFailureClassFromVerification, s as controlRunToRunRecord, t as createLlmReviewer, u as evaluateActionPolicy, v as inMemoryReviewStore, w as jsonlReviewStore, x as runProposeReview, y as runProposeReviewAsControlLoop, z as scoreFromEvals } from './control-CcBiAEnn.js';
2
+ import { R as RunRecord, b as RunSplitTag } from './run-record-B7RTi_ix.js';
3
+ export { g as AGENT_PROFILE_KINDS, h as AgentInterfaceProfileLike, A as AgentProfileCell, f as AgentProfileCellInput, i as AgentProfileCellSchemaVersion, j as AgentProfileCellValidationError, k as AgentProfileDimensionValue, l as AgentProfileHarness, a as AgentProfileJson, m as AgentProfileKind, n as AgentProfileSource, o as AgentProfileSourceInput, J as JudgeScoresRecord, c as RunCostProvenance, e as RunJudgeMetadata, p as RunOutcome, q as RunRecordValidationError, d as RunTokenUsage, r as agentProfileCellHashMaterial, s as agentProfileCellKey, t as assertRunAgentProfileCell, u as buildAgentInterfaceProfileCell, v as buildAgentProfileCell, w as groupRunsByAgentProfileCell, x as isRunRecord, y as modelHasSnapshot, z as parseRunRecordSafe, B as requireAgentProfileCell, C as resolveRunCostProvenance, D as roundTripRunRecord, E as toAgentProfileJson, F as validateAgentProfileCell, G as validateRunRecord, H as verifyAgentProfileCell } from './run-record-B7RTi_ix.js';
4
+ import { B as BehavioralMetrics, A as RunScore, a as RunTrace, E as RunScoreWeights } from './semantic-concept-judge-DxJmRkyJ.js';
5
+ export { G as ConceptComplexity, H as ConceptFinding, J as ConceptSpec, L as ConceptWeightStrategy, C as CreateAnalystAiConfig, M as DEFAULT_COMPLEXITY_WEIGHTS, N as DEFAULT_RUN_SCORE_WEIGHTS, D as DEFAULT_TRACE_ANALYST_KINDS, c as DiffPolicy, F as FAILURE_MODE_KIND_SPEC, g as FindingSubject, h as FindingSubjectKind, j as FindingsDiff, k as FindingsStore, I as IMPROVEMENT_KIND_SPEC, l as KNOWLEDGE_GAP_KIND_SPEC, m as KNOWLEDGE_POISONING_KIND_SPEC, P as PersistedFinding, R as RunCritic, O as RunCriticOptions, Q as SEMANTIC_CONCEPT_JUDGE_VERSION, n as SKILL_USAGE_ANALYST, b as SemanticConceptJudgeInput, S as SemanticConceptJudgeOptions, T as SemanticConceptJudgeResult, o as SkillUsageAnalyst, U as SuboptimalCode, V as SuboptimalSignal, W as aggregateRunScore, X as clamp01, Y as computeTraceMetrics, t as createAnalystAi, Z as createSemanticConceptJudge, u as defaultIsMaterial, v as diffFindings, _ as runSemanticConceptJudge } from './semantic-concept-judge-DxJmRkyJ.js';
6
+ import { m as ChatRequest, q as CreateChatClientOpts } from './kind-factory-DcNg13sZ.js';
7
+ export { a as Analyst, b as AnalystContext, i as AnalystCost, A as AnalystFinding, j as AnalystInputKind, k as AnalystRequirements, f as AnalystRunEvent, e as AnalystRunInputs, d as AnalystRunResult, c as AnalystRunSummary, g as AnalystSeverity, l as ChatCallOpts, C as ChatClient, n as ChatResponse, o as ChatTransport, p as CliBridgeTransportOpts, r as CreateTraceAnalystKindOpts, D as DirectProviderTransportOpts, E as EvidenceRef, M as MockTransportOpts, s as RawAnalystFinding, u as RouterTransportOpts, S as SandboxSdkTransportOpts, v as TraceAnalystGolden, T as TraceAnalystKindSpec, w as computeFindingId, x as createChatClient, y as createTraceAnalystKind, z as makeFinding, F as renderPriorFindings } from './kind-factory-DcNg13sZ.js';
8
+ export { a as AnalystHooks, A as AnalystRegistry, b as AnalystRegistryOptions, B as BudgetPolicy, D as DefaultAnalystRegistryOptions, R as RegistryRunOpts, c as buildDefaultAnalystRegistry } from './default-registry-DltpYR5u.js';
9
+ export { F as FindingToPolicyEditOptions, P as POLICY_EDIT_AXES, a as POLICY_EDIT_TARGET_SURFACES, b as PolicyEdit, c as PolicyEditAdmission, d as PolicyEditAdmissionOptions, e as PolicyEditAxis, f as PolicyEditChange, g as PolicyEditExpectedGain, h as PolicyEditGainDirection, i as PolicyEditGainUnit, j as PolicyEditInit, k as PolicyEditRisk, l as PolicyEditSchemaVersion, m as PolicyEditSource, n as PolicyEditTarget, o as PolicyEditTargetSurface, p as PolicyEditValidationError, q as admitPolicyEdit, r as applyPolicyEditToSurface, s as computePolicyEditId, t as isPolicyEdit, u as makePolicyEdit, v as policyEditFromFinding, w as policyEditsFromFindings, x as scorePolicyEditReadiness, y as validatePolicyEdit } from './policy-edit-RLn8GWof.js';
10
10
  import { TCloud } from '@tangle-network/tcloud';
11
11
  import { B as BenchmarkRunnerConfig, S as Scenario, c as BenchmarkReport, P as ProductClientConfig, C as CheckResult, T as TestResult, d as PersonaConfig, D as DriverResult, e as DriverState, b as JudgeFn, f as CollectedArtifacts, g as ScenarioResult, h as TurnMetrics, i as ScenarioFile, j as CompletionCriterion } from './types-C7DGg5ex.js';
12
12
  export { A as ArtifactCheck, k as ArtifactResult, E as EvalResult, F as FeedbackPattern, l as JudgeConfig, a as JudgeInput, m as JudgeRubric, J as JudgeScore, n as PersonaRigor, R as RouteMap, o as RubricDimension, p as Turn, q as TurnResult } from './types-C7DGg5ex.js';
@@ -16,12 +16,12 @@ import { F as FailureClass, T as ToolSpan, h as BudgetSpec, B as BudgetLedgerEnt
16
16
  export { A as Artifact, E as EventKind, i as FAILURE_CLASSES, G as GenericSpan, J as JudgeSpan, M as Message, c as RetrievalSpan, g as RunLayer, f as RunStatus, d as SandboxSpan, j as SpanBase, b as SpanKind, k as SpanStatus, l as TRACE_SCHEMA_VERSION, e as TraceEvent, m as isJudgeSpan, n as isLlmSpan, o as isRetrievalSpan, p as isSandboxSpan, q as isToolSpan } from './schema-SGWcK9wa.js';
17
17
  import { A as AgentEvalError, J as JudgeError, a as ConfigError } from './errors-oeQrLqXC.js';
18
18
  export { b as AgentEvalErrorCode, C as CaptureIntegrityError, N as NotFoundError, R as ReplayError, V as ValidationError, c as VerificationError } from './errors-oeQrLqXC.js';
19
- import { c as CorrectnessChecker } from './pre-registration-oNItiRBb.js';
20
- export { A as ArtifactCheckArtifact, e as ArtifactEventLike, f as ArtifactValidator, g as BackendIntegrityError, B as BackendIntegrityReport, h as ComparePairedArmsOptions, C as CompletionRequirement, a as CompletionVerdict, H as HypothesisManifest, i as HypothesisResult, j as LlmCorrectnessCheckerOpts, L as LlmJudgeDimension, d as LlmJudgeOptions, M as MatchedPair, k as PairArmsOptions, m as PairArmsResult, n as PairedArmRow, P as PairedArmsComparison, o as PairedCorrectness, p as PairedMetricDelta, q as ProducedProposal, b as ProducedState, r as ProposalEventLike, s as RequirementCheck, R as RuntimeEventLike, t as SatisfiedBy, S as SignedManifest, u as SignedManifestAlgo, T as TaskGold, v as ToolCallEventLike, V as ValidationContext, w as ValidationIssue, x as ValidationResult, y as assertRealBackend, z as byteLengthRange, D as canonicalize, E as comparePairedArms, F as completionVerdict, G as composeValidators, I as containsAll, J as createLlmCorrectnessChecker, K as createTokenRecallChecker, N as evaluateHypothesis, O as extractProducedState, Q as hashJson, U as jsonHasKeys, l as llmJudge, W as pairArms, X as parseCorrectnessResponse, Y as regexMatch, Z as signManifest, _ as summarizeBackendIntegrity, $ as verifyCompletion, a0 as verifyManifest } from './pre-registration-oNItiRBb.js';
19
+ import { c as CorrectnessChecker } from './pre-registration-D8h7ZxNL.js';
20
+ export { A as ArtifactCheckArtifact, e as ArtifactEventLike, f as ArtifactValidator, g as BackendIntegrityError, B as BackendIntegrityReport, h as ComparePairedArmsOptions, C as CompletionRequirement, a as CompletionVerdict, H as HypothesisManifest, i as HypothesisResult, j as LlmCorrectnessCheckerOpts, L as LlmJudgeDimension, d as LlmJudgeOptions, M as MatchedPair, k as PairArmsOptions, m as PairArmsResult, n as PairedArmRow, P as PairedArmsComparison, o as PairedCorrectness, p as PairedMetricDelta, q as ProducedProposal, b as ProducedState, r as ProposalEventLike, s as RequirementCheck, R as RuntimeEventLike, t as SatisfiedBy, S as SignedManifest, u as SignedManifestAlgo, T as TaskGold, v as ToolCallEventLike, V as ValidationContext, w as ValidationIssue, x as ValidationResult, y as assertRealBackend, z as byteLengthRange, D as canonicalize, E as comparePairedArms, F as completionVerdict, G as composeValidators, I as containsAll, J as createLlmCorrectnessChecker, K as createTokenRecallChecker, N as evaluateHypothesis, O as extractProducedState, Q as hashJson, U as jsonHasKeys, l as llmJudge, W as pairArms, X as parseCorrectnessResponse, Y as regexMatch, Z as signManifest, _ as summarizeBackendIntegrity, $ as verifyCompletion, a0 as verifyManifest } from './pre-registration-D8h7ZxNL.js';
21
21
  import { T as TraceEmitter } from './emitter-BRchAAAx.js';
22
22
  export { R as RunCompleteHook, a as RunCompleteHookContext, S as SpanHandle, b as TraceEmitterOptions, l as llmSpanFromProvider } from './emitter-BRchAAAx.js';
23
- import { h as ReleaseConfidenceThresholds, f as ReleaseConfidenceScorecard } from './release-report-oBfOz8ku.js';
24
- export { A as ActionableSideInfo, o as AsiSeverity, B as BootstrapOptions, a as BootstrapResult, J as JudgeReplayGateArgs, R as ReleaseConfidenceAxis, b as ReleaseConfidenceAxisName, c as ReleaseConfidenceInput, d as ReleaseConfidenceIssue, e as ReleaseConfidenceMetrics, g as ReleaseConfidenceStatus, i as ReleaseTraceEvidence, j as RenderReleaseReportOptions, V as Verdict, k as assertReleaseConfidence, l as bootstrapCi, m as evaluateReleaseConfidence, n as judgeReplayGate, r as renderReleaseReport } from './release-report-oBfOz8ku.js';
23
+ import { h as ReleaseConfidenceThresholds, f as ReleaseConfidenceScorecard } from './release-report-CCtzajxP.js';
24
+ export { A as ActionableSideInfo, o as AsiSeverity, B as BootstrapOptions, a as BootstrapResult, J as JudgeReplayGateArgs, R as ReleaseConfidenceAxis, b as ReleaseConfidenceAxisName, c as ReleaseConfidenceInput, d as ReleaseConfidenceIssue, e as ReleaseConfidenceMetrics, g as ReleaseConfidenceStatus, i as ReleaseTraceEvidence, j as RenderReleaseReportOptions, V as Verdict, k as assertReleaseConfidence, l as bootstrapCi, m as evaluateReleaseConfidence, n as judgeReplayGate, r as renderReleaseReport } from './release-report-CCtzajxP.js';
25
25
  export { c as CliffsMagnitude, d as CorpusAgreementOptions, e as CorpusAgreementPerDimension, C as CorpusAgreementReport, f as CorpusScoreRecord, g as EProcess, h as EProcessOptions, E as EProcessState, i as EProcessStep, M as McNemarResult, P as PairedBootstrapOptions, a as PairedBootstrapResult, j as PairedSignTestResult, k as ProportionInterval, R as RiskDifferenceResult, S as SignTestAlternative, W as WeightedCompositeInput, l as WeightedCompositeResult, b as benjaminiHochberg, m as bonferroni, n as cliffsDelta, o as cohensD, q as confidenceInterval, r as corpusInterRaterAgreement, s as corpusInterRaterAgreementFromJudgeScores, t as eProcess, u as holm, v as interRaterReliability, x as interpretCliffs, y as mannWhitneyU, z as mcnemar, A as mcnemarPower, B as mcnemarRequiredN, D as mulberry32, F as normalizeScores, p as pairedBootstrap, G as pairedMde, H as pairedRiskDifference, I as pairedSignTest, J as pairedTTest, K as partialCredit, L as passAtK, N as pearsonR, O as ranks, Q as requiredSampleSize, T as spearmanR, U as weightedComposite, V as weightedMean, w as wilcoxonSignedRank, X as wilson } from './statistics-oUbOJe-S.js';
26
26
  import { OtelExporter, OtelExportConfig } from './traces.js';
27
27
  export { CaptureFetchContext, CaptureFetchOptions, DEFAULT_REDACTION_RULES, ExportableSpan, ExtractedUsage, FlattenOtlpOptions, LLM_CACHED_TOKENS, LLM_CACHED_TOKEN_ATTR_KEYS, LLM_COST_ATTR_KEYS, LLM_COST_USD, LLM_INPUT_TOKENS, LLM_INPUT_TOKEN_ATTR_KEYS, LLM_MODEL_ATTR_KEYS, LLM_MODEL_NAME, LLM_OUTPUT_TOKENS, LLM_OUTPUT_TOKEN_ATTR_KEYS, OPENINFERENCE_SPAN_KIND, OTEL_AGENT_EVAL_SCOPE, OtlpExport, OtlpFileTraceStore, OtlpFileTraceStoreOptions, OtlpFlatLine, OtlpResourceSpans, OtlpSpan, OtlpToRunRecordsOptions, OtlpTraceRunRecord, ProjectedOtlpSpan, REDACTION_VERSION, RedactionReport, RedactionRule, ReplayCache, ReplayCacheEntry, ReplayCacheMissError, ReplayCacheStats, ReplayFetchOptions, SPAN_KIND_ATTR_KEYS, SpanNotFoundError, TOOL_NAME, TOOL_NAME_ATTR_KEYS, TRACE_ANALYST_ACTOR_DESCRIPTION, TRACE_ANALYST_ACTOR_DESCRIPTION_VERSION, TRACE_ANALYST_SUBAGENT_DESCRIPTION, TraceAggregate, TraceAnalystHookOptions, TraceFileMissingError, TraceInsightContext, TraceInsightFinding, TraceInsightPanelRole, TraceInsightPromptInput, TraceInsightQualityGate, TraceInsightQuestion, TraceInsightReadiness, TraceInsightSuite, TraceInsightTask, TraceNotFoundError, TraceStoreSource, TraceStoreToOtlpOptions, TracesToOtlpResult, asNumber, asString, buildTraceAnalystTools, buildTraceInsightContext, buildTraceInsightPrompt, captureFetchToRawSink, convertTraceStoresToOtlp, createOtelExporter, createOtelTracingStore, createReplayFetch, defaultTraceInsightPanel, describeTraceInsightScope, domainEvidencePattern, exportRunAsOtlp, extractOtlpAttributes, extractUsage, extractUsageFromResponse, extractUsageFromSse, firstNumberAttr, firstStringAttr, flattenOtlpExportToNdjson, inferDomainKeywords, inferOtlpKind, iterateRawCalls, otelRunCompleteHook, otlpToRunRecords, otlpToTraceRunRecords, planTraceInsightQuestions, projectOtlpFlatLine, readOtlpStatus, redactString, redactValue, scoreTraceInsightReadiness, stringField, tokenizeDomainWords, traceAnalystFunctionGroup, traceAnalystOnRunComplete, traceSpanKindToOpenInferenceKind } from './traces.js';
@@ -29,10 +29,10 @@ import { a as AnalyzeTracesInput, A as AnalyzeTracesOptions, b as AnalyzeTracesR
29
29
  export { c as AnalyzeTracesTurnSnapshot, d as analyzeTraces } from './analyst-C8HHvfJp.js';
30
30
  import { a as TraceAnalystSpan } from './store-C1YxJDEK.js';
31
31
  export { D as DEFAULT_TRACE_ANALYST_BUDGETS, b as DatasetOverview, E as ErrorCluster, Q as QueryTracesPage, S as SearchSpanResult, c as SearchTraceResult, d as SpanMatchRecord, e as TRACE_ANALYST_TRUNCATION_MARKER_PREFIX, T as TraceAnalysisStore, f as TraceAnalystByteBudgets, g as TraceAnalystFilters, h as TraceAnalystSpanKind, i as TraceAnalystSpanStatus, j as TraceAnalystTraceSummary, V as ViewSpansResult, k as ViewTraceOversized, l as ViewTraceResult } from './store-C1YxJDEK.js';
32
- import { b as JudgeConfig, J as JudgeScore, S as Scenario$1, g as Gate } from './types-CgSlO6wT.js';
33
- import { A as AnalyzeRunsOptions } from './analyze-runs-Dmz6LA9e.js';
34
- import { q as Objective, s as ParetoResult, h as GepaProposerConstraints, a as RunImprovementLoopResult } from './gepa-DolL_Fko.js';
35
- export { t as DEFAULT_RED_TEAM_CORPUS, D as Direction, e as RedTeamCase, u as RedTeamCategory, v as RedTeamFinding, w as RedTeamPayload, x as RedTeamReport, y as crowdingDistance, z as dominates, A as paretoFrontier, B as paretoFrontierWithCrowding, E as redTeamDataset, F as redTeamReport, H as scalarScore, I as scoreRedTeamOutput, J as toolNamesForRun } from './gepa-DolL_Fko.js';
32
+ import { b as JudgeConfig, J as JudgeScore, S as Scenario$1, g as Gate } from './types-C5gJrOVT.js';
33
+ import { A as AnalyzeRunsOptions } from './analyze-runs-BYHg6Irm.js';
34
+ import { q as Objective, s as ParetoResult, h as GepaProposerConstraints, a as RunImprovementLoopResult } from './gepa-dne9JDPL.js';
35
+ export { t as DEFAULT_RED_TEAM_CORPUS, D as Direction, e as RedTeamCase, u as RedTeamCategory, v as RedTeamFinding, w as RedTeamPayload, x as RedTeamReport, y as crowdingDistance, z as dominates, A as paretoFrontier, B as paretoFrontierWithCrowding, E as redTeamDataset, F as redTeamReport, H as scalarScore, I as scoreRedTeamOutput, J as toolNamesForRun } from './gepa-dne9JDPL.js';
36
36
  import { S as SandboxDriver, H as HarnessConfig, a as SandboxHarnessResult } from './test-graded-scenario-mzYBKspu.js';
37
37
  export { D as DockerSandboxDriver, c as SandboxHarness, d as SandboxResult, e as SubprocessSandboxDriver, f as SubprocessSandboxDriverOptions, g as TestGradedRunOptions, b as TestGradedRunResult, T as TestGradedScenario, h as TestOutputParser, i as composeParsers, j as jestTestParser, p as pytestTestParser, r as runTestGradedScenario, v as vitestTestParser } from './test-graded-scenario-mzYBKspu.js';
38
38
  export { b as RunIntegrityError, R as RunIntegrityExpectations, c as RunIntegrityIssue, d as RunIntegrityIssueCode, a as RunIntegrityReport, e as assertRunCaptured, t as throwIfRunIncomplete } from './integrity-qemeBAyx.js';
@@ -41,7 +41,7 @@ export { F as FileSystemRawProviderSink, a as FileSystemRawProviderSinkOptions,
41
41
  import { T as TraceStore, R as RunFilter } from './store-BsVi7ncX.js';
42
42
  export { E as EventFilter, F as FileSystemTraceStore, a as FileSystemTraceStoreOptions, I as InMemoryTraceStore, S as SpanFilter } from './store-BsVi7ncX.js';
43
43
  export { D as DEFAULT_FAILURE_RULES, b as FailureClassification, c as FailureContext, d as FailureRule, e as classifyFailure } from './failure-cluster-C48PiReX.js';
44
- export { P as ProjectRuntimeTrajectoryEvidenceOptions, a as RuntimeTrajectoryEvidenceProjection, b as RuntimeTrajectoryEvidenceSummary, c as RuntimeTrajectoryHookEvent, R as RuntimeTrajectoryRecord, d as RuntimeTrajectoryRunRecord, p as parseRuntimeTrajectoryHookEvent, e as projectRuntimeTrajectoryEvidence } from './runtime-trajectory-h5i0SZUj.js';
44
+ export { P as ProjectRuntimeTrajectoryEvidenceOptions, a as RuntimeTrajectoryEvidenceProjection, b as RuntimeTrajectoryEvidenceSummary, c as RuntimeTrajectoryHookEvent, R as RuntimeTrajectoryRecord, d as RuntimeTrajectoryRunRecord, p as parseRuntimeTrajectoryHookEvent, e as projectRuntimeTrajectoryEvidence } from './runtime-trajectory-Dws7Kpgi.js';
45
45
  import { a as BaselineReport, b as Trajectory, T as TrajectoryStep } from './baseline-DsNteOgR.js';
46
46
  export { B as BaselineOptions, M as MetricSamples, d as MetricVerdict, e as ToolStats, f as ToolUseMetrics, g as ToolUseOptions, h as buildTrajectory, i as compareToBaseline, c as computeToolUseMetrics, j as iqr, w as welchsTTest } from './baseline-DsNteOgR.js';
47
47
  import { HarnessType, AgentProfile } from '@tangle-network/agent-interface';
@@ -57,13 +57,13 @@ import { L as Layer, S as Severity, b as LayerResult, c as VerifyContext } from
57
57
  export { F as Finding, d as LayerStatus, M as MultiLayerVerifier, a as VerificationReport, V as VerifyOptions, g as gradeSemanticStatus } from './multi-layer-verifier-BsqKuLyN.js';
58
58
  import { L as LlmClientOptions } from './llm-client-DyqEH4jH.js';
59
59
  export { d as LlmCallError, b as LlmCallRequest, c as LlmCallResult, e as LlmClient, f as LlmMessage, g as LlmRouteAssertionError, a as LlmRouteRequirements, h as LlmUsage, i as assertLlmRoute, j as backoffMs, k as callLlm, l as callLlmJson, m as isTransientLlmError, p as probeLlm, s as stripFencedJson } from './llm-client-DyqEH4jH.js';
60
- export { B as BENCHMARK_SPLIT_SEED, a as BenchmarkAdapter, b as BenchmarkDatasetItem, c as BenchmarkEvaluation, d as BenchmarkFamily, e as BenchmarkResponder, f as BenchmarkScenario, g as BenchmarkSource, h as BenchmarkTaskKind, i as benchmarkDeterministicSplit, j as benchmarks } from './index-CWr5SIG-.js';
61
- export { C as CallbackResearcher, d as CallbackResearcherOptions, e as CampaignFactoryParams, f as CampaignIntegrityPolicy, g as CampaignRunContext, h as CampaignRunOutcome, i as CampaignRunner, j as CampaignScenario, k as CampaignVariant, c as EvalCampaignOptions, b as EvalCampaignResult, E as ExperimentPlan, a as ExperimentResult, l as FailedRun, F as FailureMode, N as NoopResearcher, R as Researcher, S as SteeringChange, r as runEvalCampaign } from './researcher-CaH0CwFC.js';
62
- export { G as GainDistributionBin, a as GainDistributionFigureSpec, b as GainDistributionOptions, m as GateDecision, n as GateEvidence, H as HeldOutGate, o as HeldOutGateConfig, q as HeldOutGateRejectionCode, P as ParetoFigureSpec, c as ParetoPoint, R as RESEARCH_REPORT_HARD_PAIR_FLOOR, d as ResearchReport, e as ResearchReportCandidate, f as ResearchReportDecision, g as ResearchReportMethodology, h as ResearchReportOptions, i as ResearchReportRecommendation, S as SummaryTable, j as SummaryTableOptions, k as SummaryTableRow, l as gainHistogram, p as paretoChart, r as researchReport, s as summaryTable } from './summary-report-Bz-0-t8v.js';
60
+ export { B as BENCHMARK_SPLIT_SEED, a as BenchmarkAdapter, b as BenchmarkDatasetItem, c as BenchmarkEvaluation, d as BenchmarkFamily, e as BenchmarkResponder, f as BenchmarkScenario, g as BenchmarkSource, h as BenchmarkTaskKind, i as benchmarkDeterministicSplit, j as benchmarks } from './index-BTEpx9He.js';
61
+ export { C as CallbackResearcher, d as CallbackResearcherOptions, e as CampaignFactoryParams, f as CampaignIntegrityPolicy, g as CampaignRunContext, h as CampaignRunOutcome, i as CampaignRunner, j as CampaignScenario, k as CampaignVariant, c as EvalCampaignOptions, b as EvalCampaignResult, E as ExperimentPlan, a as ExperimentResult, l as FailedRun, F as FailureMode, N as NoopResearcher, R as Researcher, S as SteeringChange, r as runEvalCampaign } from './researcher-Dq-EtpbE.js';
62
+ export { G as GainDistributionBin, a as GainDistributionFigureSpec, b as GainDistributionOptions, m as GateDecision, n as GateEvidence, H as HeldOutGate, o as HeldOutGateConfig, q as HeldOutGateRejectionCode, P as ParetoFigureSpec, c as ParetoPoint, R as RESEARCH_REPORT_HARD_PAIR_FLOOR, d as ResearchReport, e as ResearchReportCandidate, f as ResearchReportDecision, g as ResearchReportMethodology, h as ResearchReportOptions, i as ResearchReportRecommendation, S as SummaryTable, j as SummaryTableOptions, k as SummaryTableRow, l as gainHistogram, p as paretoChart, r as researchReport, s as summaryTable } from './summary-report-BJ5aNwZ1.js';
63
63
  export { I as InterimReleaseConfidence, a as InterimReleaseConfidenceInput, P as PairedEvalueOptions, b as PairedEvalueSequence, c as PairedEvalueStep, S as SequentialDecision, e as evaluateInterimReleaseConfidence, p as pairedEvalueSequence } from './sequential-5iSVfzl2.js';
64
64
  import '@ax-llm/ax';
65
65
  import 'zod';
66
- import './insight-report-D4cXFsLt.js';
66
+ import './insight-report-IwwvqZZv.js';
67
67
  import './storage-Dw_f7WMt.js';
68
68
 
69
69
  /**
package/dist/index.js CHANGED
@@ -9,7 +9,7 @@ import {
9
9
  checkBehavioralCanary,
10
10
  checkCanaries,
11
11
  runBehavioralCanaries
12
- } from "./chunk-WBOGKYM4.js";
12
+ } from "./chunk-J6P6PK2R.js";
13
13
  import {
14
14
  BENCHMARK_SPLIT_SEED,
15
15
  benchmarks_exports,
@@ -61,7 +61,7 @@ import {
61
61
  pairArms,
62
62
  parseCorrectnessResponse,
63
63
  verifyCompletion
64
- } from "./chunk-KDCMEZDI.js";
64
+ } from "./chunk-KG4TD7EQ.js";
65
65
  import {
66
66
  DEFAULT_MUTATION_PRIMITIVES,
67
67
  DEFAULT_RED_TEAM_CORPUS,
@@ -85,7 +85,7 @@ import {
85
85
  scoreRedTeamOutput,
86
86
  surfaceContentHash,
87
87
  toolNamesForRun
88
- } from "./chunk-N6MTC3GK.js";
88
+ } from "./chunk-ADYLPOSX.js";
89
89
  import {
90
90
  MODEL_PRICING,
91
91
  MetricsCollector,
@@ -119,11 +119,11 @@ import {
119
119
  defaultIsMaterial,
120
120
  diffFindings,
121
121
  runSemanticConceptJudge
122
- } from "./chunk-I2HNIE6N.js";
122
+ } from "./chunk-WSBUZMBU.js";
123
123
  import {
124
124
  buildDefaultAnalystRegistry,
125
125
  computeTraceMetrics
126
- } from "./chunk-LVTGFSHF.js";
126
+ } from "./chunk-I6LVHOV3.js";
127
127
  import {
128
128
  DEFAULT_RUN_SCORE_WEIGHTS,
129
129
  Mutex,
@@ -141,7 +141,7 @@ import {
141
141
  policyEditsFromFindings,
142
142
  scorePolicyEditReadiness,
143
143
  validatePolicyEdit
144
- } from "./chunk-AN5UYSVD.js";
144
+ } from "./chunk-QMXXSNC4.js";
145
145
  import {
146
146
  AnalystRegistry,
147
147
  DEFAULT_TRACE_ANALYST_KINDS,
@@ -153,7 +153,7 @@ import {
153
153
  createTraceAnalystKind,
154
154
  makeFinding,
155
155
  renderPriorFindings
156
- } from "./chunk-DWLIGZBX.js";
156
+ } from "./chunk-5S5NJ63F.js";
157
157
  import {
158
158
  allCriticalPassed,
159
159
  controlFailureClassFromVerification,
@@ -174,7 +174,7 @@ import {
174
174
  stopOnNoProgress,
175
175
  stopOnRepeatedAction,
176
176
  subjectiveEval
177
- } from "./chunk-IMWDSFUM.js";
177
+ } from "./chunk-DXZRATT5.js";
178
178
  import {
179
179
  assertReleaseConfidence,
180
180
  bootstrapCi,
@@ -184,7 +184,7 @@ import {
184
184
  } from "./chunk-MOXWMGPC.js";
185
185
  import {
186
186
  runEvalCampaign
187
- } from "./chunk-LOBMT6SB.js";
187
+ } from "./chunk-ONM6PEAE.js";
188
188
  import "./chunk-ARU2PZFM.js";
189
189
  import {
190
190
  evaluateInterimReleaseConfidence,
@@ -267,7 +267,7 @@ import {
267
267
  scoreTraceInsightReadiness,
268
268
  tokenizeDomainWords,
269
269
  traceAnalystOnRunComplete
270
- } from "./chunk-RSVSSZKF.js";
270
+ } from "./chunk-TLDB7WRY.js";
271
271
  import {
272
272
  FAILURE_CLASSES,
273
273
  TRACE_SCHEMA_VERSION,
@@ -345,9 +345,10 @@ import {
345
345
  isRunRecord,
346
346
  modelHasSnapshot,
347
347
  parseRunRecordSafe,
348
+ resolveRunCostProvenance,
348
349
  roundTripRunRecord,
349
350
  validateRunRecord
350
- } from "./chunk-VK6HBGAE.js";
351
+ } from "./chunk-5UF54T55.js";
351
352
  import {
352
353
  AGENT_PROFILE_KINDS,
353
354
  AgentProfileCellValidationError,
@@ -380,7 +381,7 @@ import {
380
381
  isTransientLlmError,
381
382
  probeLlm,
382
383
  stripFencedJson
383
- } from "./chunk-FUCQVFMU.js";
384
+ } from "./chunk-GY4SYVPJ.js";
384
385
  import {
385
386
  FileSystemRawProviderSink,
386
387
  InMemoryRawProviderSink,
@@ -11924,6 +11925,7 @@ export {
11924
11925
  requiredSampleSize,
11925
11926
  researchReport,
11926
11927
  resolveModelPricing,
11928
+ resolveRunCostProvenance,
11927
11929
  resolveSeat,
11928
11930
  roundTripRunRecord,
11929
11931
  routeFields,