@tangle-network/agent-eval 0.145.6 → 0.145.7
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +10 -0
- package/README.md +4 -0
- package/dist/analyst/index.d.ts +10 -10
- package/dist/analyst/index.js +1 -1
- package/dist/{backend-integrity-CsVin_Wb.d.ts → backend-integrity-BffcHGdm.d.ts} +2 -2
- package/dist/{backend-integrity-CsVin_Wb.d.ts.map → backend-integrity-BffcHGdm.d.ts.map} +1 -1
- package/dist/{benchmark-Ceoan7vk.d.ts → benchmark-DbiZbPdH.d.ts} +3 -3
- package/dist/{benchmark-Ceoan7vk.d.ts.map → benchmark-DbiZbPdH.d.ts.map} +1 -1
- package/dist/{benchmark-command-DspwA7cv.js → benchmark-command-C7L0rQsD.js} +2 -2
- package/dist/{benchmark-command-DspwA7cv.js.map → benchmark-command-C7L0rQsD.js.map} +1 -1
- package/dist/benchmarks/index.d.ts +5 -5
- package/dist/benchmarks/index.js +2 -2
- package/dist/campaign/index.d.ts +7 -7
- package/dist/campaign/index.js +3 -3
- package/dist/{campaign-jTOvqnse.js → campaign-DRcGtxEf.js} +35 -11
- package/dist/campaign-DRcGtxEf.js.map +1 -0
- package/dist/{capture-fetch-DDvpjVRU.d.ts → capture-fetch-hykwHfDI.d.ts} +2 -2
- package/dist/{capture-fetch-DDvpjVRU.d.ts.map → capture-fetch-hykwHfDI.d.ts.map} +1 -1
- package/dist/cli.js +1 -1
- package/dist/{client-DCVe0CwG.d.ts → client-D_TIV9pJ.d.ts} +4 -4
- package/dist/{client-DCVe0CwG.d.ts.map → client-D_TIV9pJ.d.ts.map} +1 -1
- package/dist/contract/index.d.ts +11 -11
- package/dist/contract/index.js +4 -4
- package/dist/{default-registry-CgzxnJsj.d.ts → default-registry-cpHtP2Og.d.ts} +6 -6
- package/dist/{default-registry-CgzxnJsj.d.ts.map → default-registry-cpHtP2Og.d.ts.map} +1 -1
- package/dist/{define-agent-eval-DLhZ8IHi.d.ts → define-agent-eval-DBOpAM_g.d.ts} +6 -6
- package/dist/{define-agent-eval-DLhZ8IHi.d.ts.map → define-agent-eval-DBOpAM_g.d.ts.map} +1 -1
- package/dist/{define-agent-eval-CViDh2P9.js → define-agent-eval-m5lM7XGE.js} +4 -4
- package/dist/{define-agent-eval-CViDh2P9.js.map → define-agent-eval-m5lM7XGE.js.map} +1 -1
- package/dist/{engine-FRZ3RLvT.d.ts → engine-DJqRKbhs.d.ts} +7 -7
- package/dist/{engine-FRZ3RLvT.d.ts.map → engine-DJqRKbhs.d.ts.map} +1 -1
- package/dist/{eval-campaign-UB-usSQ2.js → eval-campaign-aNqpefCS.js} +2 -2
- package/dist/{eval-campaign-UB-usSQ2.js.map → eval-campaign-aNqpefCS.js.map} +1 -1
- package/dist/{exact-types-BH1twmAJ.d.ts → exact-types-CzbhVDr2.d.ts} +2 -2
- package/dist/{exact-types-BH1twmAJ.d.ts.map → exact-types-CzbhVDr2.d.ts.map} +1 -1
- package/dist/experiment/index.d.ts +3 -3
- package/dist/{feedback-trajectory-juOozjAc.d.ts → feedback-trajectory-D9vYSob_.d.ts} +3 -3
- package/dist/{feedback-trajectory-juOozjAc.d.ts.map → feedback-trajectory-D9vYSob_.d.ts.map} +1 -1
- package/dist/hosted/index.d.ts +3 -3
- package/dist/index-BKShPTwZ.d.ts +1 -0
- package/dist/{index-COQYtuRF.d.ts → index-BNPtkBPf.d.ts} +2 -2
- package/dist/{index-COQYtuRF.d.ts.map → index-BNPtkBPf.d.ts.map} +1 -1
- package/dist/{index-DF3ynqmJ.d.ts → index-D7gl974A.d.ts} +11 -11
- package/dist/{index-DF3ynqmJ.d.ts.map → index-D7gl974A.d.ts.map} +1 -1
- package/dist/{index-PgqfwhsM.d.ts → index-YrUFx3FU.d.ts} +5 -5
- package/dist/{index-PgqfwhsM.d.ts.map → index-YrUFx3FU.d.ts.map} +1 -1
- package/dist/index.d.ts +22 -22
- package/dist/index.js +9 -9
- package/dist/{insight-report-CRi-Ufrj.d.ts → insight-report-BeT8KCgI.d.ts} +3 -3
- package/dist/{insight-report-CRi-Ufrj.d.ts.map → insight-report-BeT8KCgI.d.ts.map} +1 -1
- package/dist/{llm-judge-DuYa4SEA.js → llm-judge-BO1LGjdn.js} +2 -2
- package/dist/{llm-judge-DuYa4SEA.js.map → llm-judge-BO1LGjdn.js.map} +1 -1
- package/dist/meta-eval/index.d.ts +1 -1
- package/dist/{mint-Dj9Ww_3I.js → mint-BV6tLVWl.js} +2 -2
- package/dist/{mint-Dj9Ww_3I.js.map → mint-BV6tLVWl.js.map} +1 -1
- package/dist/multishot/index.d.ts +1 -1
- package/dist/openapi.json +1 -1
- package/dist/{pre-registration-DeRvl9sE.d.ts → pre-registration-zFSLEiFU.d.ts} +2 -2
- package/dist/{pre-registration-DeRvl9sE.d.ts.map → pre-registration-zFSLEiFU.d.ts.map} +1 -1
- package/dist/{produced-state-D6j7qy1Q.js → produced-state-C6vSgOpE.js} +2 -2
- package/dist/{produced-state-D6j7qy1Q.js.map → produced-state-C6vSgOpE.js.map} +1 -1
- package/dist/{promotion-policy-CsMZJOB-.d.ts → promotion-policy-u3wj6w3U.d.ts} +2 -2
- package/dist/{promotion-policy-CsMZJOB-.d.ts.map → promotion-policy-u3wj6w3U.d.ts.map} +1 -1
- package/dist/{provenance-8L-_4xiL.d.ts → provenance-BOtMtoiZ.d.ts} +5 -5
- package/dist/{provenance-8L-_4xiL.d.ts.map → provenance-BOtMtoiZ.d.ts.map} +1 -1
- package/dist/{registry-oJeeI4-a.d.ts → registry-B_1Frl8a.d.ts} +3 -3
- package/dist/{registry-oJeeI4-a.d.ts.map → registry-B_1Frl8a.d.ts.map} +1 -1
- package/dist/{release-confidence-BFRE5WSp.d.ts → release-confidence-4XrqlpFD.d.ts} +3 -3
- package/dist/{release-confidence-BFRE5WSp.d.ts.map → release-confidence-4XrqlpFD.d.ts.map} +1 -1
- package/dist/{release-confidence-CxDuiAev.js → release-confidence-BknrpBnO.js} +2 -2
- package/dist/{release-confidence-CxDuiAev.js.map → release-confidence-BknrpBnO.js.map} +1 -1
- package/dist/reporting.d.ts +3 -3
- package/dist/reporting.js +1 -1
- package/dist/{researcher-DaN4GST-.d.ts → researcher-Du-oniHp.d.ts} +3 -3
- package/dist/{researcher-DaN4GST-.d.ts.map → researcher-Du-oniHp.d.ts.map} +1 -1
- package/dist/{reward-hacking-DSSTuI9r.d.ts → reward-hacking-Bu8ev6PR.d.ts} +2 -2
- package/dist/{reward-hacking-DSSTuI9r.d.ts.map → reward-hacking-Bu8ev6PR.d.ts.map} +1 -1
- package/dist/{reward-hacking-DNgjilrV.js → reward-hacking-DFo2FU5J.js} +2 -2
- package/dist/{reward-hacking-DNgjilrV.js.map → reward-hacking-DFo2FU5J.js.map} +1 -1
- package/dist/rl.d.ts +5 -5
- package/dist/rl.js +4 -4
- package/dist/rollout/index.d.ts +1 -1
- package/dist/rollout/index.js +2 -2
- package/dist/{rollout-BWtw0I_6.js → rollout-ytVQ7WT8.js} +2 -2
- package/dist/{rollout-BWtw0I_6.js.map → rollout-ytVQ7WT8.js.map} +1 -1
- package/dist/{rubric-predictive-validity-BgxtKe4G.d.ts → rubric-predictive-validity-C7LnNvF2.d.ts} +2 -2
- package/dist/{rubric-predictive-validity-BgxtKe4G.d.ts.map → rubric-predictive-validity-C7LnNvF2.d.ts.map} +1 -1
- package/dist/{run-record-BvHPVS-i.js → run-record-D2lDdSAz.js} +5 -3
- package/dist/run-record-D2lDdSAz.js.map +1 -0
- package/dist/{run-record-CKiihE6f.d.ts → run-record-DVV82Gwh.d.ts} +8 -3
- package/dist/run-record-DVV82Gwh.d.ts.map +1 -0
- package/dist/{skillopt-optimization-method-CFDSzD-7.d.ts → skillopt-optimization-method-BPQwXkkY.d.ts} +5 -5
- package/dist/{skillopt-optimization-method-CFDSzD-7.d.ts.map → skillopt-optimization-method-BPQwXkkY.d.ts.map} +1 -1
- package/dist/{skillopt-optimization-method-BTls2l-Z.js → skillopt-optimization-method-Ceic2bfU.js} +2 -2
- package/dist/{skillopt-optimization-method-BTls2l-Z.js.map → skillopt-optimization-method-Ceic2bfU.js.map} +1 -1
- package/dist/{statistical-heldout-C4De2tRI.d.ts → statistical-heldout-CBGPSZ5X.d.ts} +3 -3
- package/dist/{statistical-heldout-C4De2tRI.d.ts.map → statistical-heldout-CBGPSZ5X.d.ts.map} +1 -1
- package/dist/{store-tool-spans-CggeC1LB.d.ts → store-tool-spans-C9c4R7ca.d.ts} +4 -4
- package/dist/{store-tool-spans-CggeC1LB.d.ts.map → store-tool-spans-C9c4R7ca.d.ts.map} +1 -1
- package/dist/{summary-report-CaL-Hnxt.d.ts → summary-report-B__Y5ub3.d.ts} +2 -2
- package/dist/{summary-report-CaL-Hnxt.d.ts.map → summary-report-B__Y5ub3.d.ts.map} +1 -1
- package/dist/{tool-groups-BdcoEgtV.d.ts → tool-groups-CZotz-e_.d.ts} +3 -3
- package/dist/tool-groups-CZotz-e_.d.ts.map +1 -0
- package/dist/trace-repair/index.d.ts +2 -2
- package/dist/traces.d.ts +6 -6
- package/dist/traces.js +1 -1
- package/dist/{types-BxLccGMf.d.ts → types-BjsNDR49.d.ts} +3 -3
- package/dist/{types-BxLccGMf.d.ts.map → types-BjsNDR49.d.ts.map} +1 -1
- package/dist/{types-BRsxjg7z.d.ts → types-C4bSVIr7.d.ts} +3 -3
- package/dist/{types-BRsxjg7z.d.ts.map → types-C4bSVIr7.d.ts.map} +1 -1
- package/dist/{types-DLQx4mKU.d.ts → types-vXyshMwx.d.ts} +2 -2
- package/dist/{types-DLQx4mKU.d.ts.map → types-vXyshMwx.d.ts.map} +1 -1
- package/dist/wire/index.d.ts +1 -1
- package/docs/eval-surface-map.md +2 -1
- package/docs/wire-protocol.md +4 -0
- package/package.json +1 -1
- package/dist/campaign-jTOvqnse.js.map +0 -1
- package/dist/index-Ba3YrbAL.d.ts +0 -1
- package/dist/run-record-BvHPVS-i.js.map +0 -1
- package/dist/run-record-CKiihE6f.d.ts.map +0 -1
- package/dist/tool-groups-BdcoEgtV.d.ts.map +0 -1
package/dist/index.d.ts
CHANGED
|
@@ -2,49 +2,49 @@ import { a as JudgeError, c as ValidationError, i as ConfigError, n as AgentEval
|
|
|
2
2
|
import { C as verifyAgentProfileCell, h as agentProfileCellKey, i as AgentProfileCellInput, l as AgentProfileJson, m as agentProfileCellHashMaterial, p as AgentProfileSourceInput, r as AgentProfileCell, s as AgentProfileDimensionValue, t as AGENT_PROFILE_KINDS, v as buildAgentProfileCell, x as toAgentProfileJson, y as groupRunsByAgentProfileCell } from "./agent-profile-cell-BkcRDikH.js";
|
|
3
3
|
import { C as defineEquivalenceCheck, S as buildEquivalenceRecord, _ as StrategyChecker, a as CheckerIdentity, b as VerificationStrategyProfile, c as EquivalenceCheckDefinition, d as EquivalenceCheckerInput, f as EquivalenceCheckerResult, g as EquivalenceRecord, h as EquivalenceProtocolError, i as equivalenceVerdict, l as EquivalenceCheckSpec, m as EquivalenceObligationStatus, n as VerdictCertification, o as CheckerOutcome, p as EquivalenceObligation, r as certificationEvidenceDigest, s as EquivalenceArm, t as DefaultVerdict, u as EquivalenceChecker, v as VERIFICATION_STRATEGIES, w as runEquivalenceCheck, x as VerificationStrategySource, y as VERIFICATION_STRATEGY_SOURCES } from "./verdict-E4eRNf7-.js";
|
|
4
4
|
import { a as MultiLayerVerifier, c as VerifyOptions, i as LayerStatus, l as gradeSemanticStatus, n as Layer, o as Severity, r as LayerResult, s as VerificationReport, t as Finding } from "./multi-layer-verifier-DIguZc8Z.js";
|
|
5
|
-
import { A as SemanticConceptJudgeInput, D as ConceptFinding, E as createDspyRlmTraceEngine, F as RunScoreWeights, I as aggregateRunScore, L as clamp01, M as SemanticConceptJudgeResult, N as runSemanticConceptJudge, O as ConceptSpec, P as RunScore, T as DspyRlmTraceEngineOptions, a as FAILURE_MODE_KIND_SPEC, d as FindingsStore, f as PersistedFinding, j as SemanticConceptJudgeOptions, k as SEMANTIC_CONCEPT_JUDGE_VERSION, l as DiffPolicy, m as diffFindings, t as DEFAULT_TRACE_ANALYST_KINDS, u as FindingsDiff, v as FindingSubject, y as FindingSubjectKind } from "./index-
|
|
5
|
+
import { A as SemanticConceptJudgeInput, D as ConceptFinding, E as createDspyRlmTraceEngine, F as RunScoreWeights, I as aggregateRunScore, L as clamp01, M as SemanticConceptJudgeResult, N as runSemanticConceptJudge, O as ConceptSpec, P as RunScore, T as DspyRlmTraceEngineOptions, a as FAILURE_MODE_KIND_SPEC, d as FindingsStore, f as PersistedFinding, j as SemanticConceptJudgeOptions, k as SEMANTIC_CONCEPT_JUDGE_VERSION, l as DiffPolicy, m as diffFindings, t as DEFAULT_TRACE_ANALYST_KINDS, u as FindingsDiff, v as FindingSubject, y as FindingSubjectKind } from "./index-YrUFx3FU.js";
|
|
6
6
|
import { C as PendingCostCall, D as costForUsage, E as costForTokenPricing, O as modelPriceKey, S as PaidCallResult, T as RunPaidCallInput, _ as CostReservationExceededError, a as CostChannel, b as CustomTokenPricing, c as CostLedgerHandle, d as CostLedgerPersistenceError, f as CostLedgerSummary, g as CostReceiptInput, h as CostReceiptCaptureError, i as CostCeilingReachedError, l as CostLedgerOptions, m as CostReceipt, n as CostAccountingIncompleteError, o as CostLedger, p as CostProvenance, r as CostCallConflictError, s as CostLedgerFilter, t as ChannelRollup, u as CostLedgerPersistence, v as CostResult, w as PendingCostCallView, x as MaximumCharge, y as CostUsage } from "./cost-ledger-DbQdN3nO.js";
|
|
7
7
|
import { C as TraceEvent, O as isToolSpan, S as ToolSpan, T as isLlmSpan, _ as Span, a as FAILURE_CLASSES, c as JudgeSpan, f as Run, h as RunStatus, l as LlmSpan, n as BudgetLedgerEntry, o as FailureClass, r as BudgetSpec, s as GenericSpan, t as Artifact, w as isJudgeSpan } from "./schema-BtVldJ3T.js";
|
|
8
|
-
import { a as RunRecord, c as RunTaskFailure, d as
|
|
9
|
-
import { a as ExtractedUsage, l as extractUsageFromSse, r as captureFetchToRawSink, s as extractUsage } from "./capture-fetch-
|
|
8
|
+
import { _ as validateRunRecord, a as RunRecord, c as RunTaskFailure, d as UNKNOWN_MODEL, f as isRunRecord, g as runTaskScore, h as roundTripRunRecord, i as RunOutcome, l as RunTerminalOutcome, m as parseRunRecordSafe, n as RunCostProvenance, o as RunRecordValidationError, p as modelHasSnapshot, r as RunJudgeMetadata, s as RunSplitTag, t as JudgeScoresRecord, u as RunTokenUsage } from "./run-record-DVV82Gwh.js";
|
|
9
|
+
import { a as ExtractedUsage, l as extractUsageFromSse, r as captureFetchToRawSink, s as extractUsage } from "./capture-fetch-hykwHfDI.js";
|
|
10
10
|
import { A as LlmClientOptions, B as isTransientLlmError, D as LlmCallRequest, E as LlmCallMetadata, F as assertLlmRoute, G as AssertCrossFamilyOptions, H as probeLlm, I as callLlm, J as assertCrossFamily, K as CrossFamilyError, L as callLlmJson, M as LlmResponseError, N as LlmRouteRequirements, O as LlmCallResult, P as LlmUsage, Q as InMemoryRawProviderSink, R as costReceiptFromLlm, T as LlmCallError, U as stripFencedJson, V as maximumChargeForLlmRequest, W as ServedModelCheck, X as FileSystemRawProviderSink, Y as judgeFamily, c as PersonaConfig, d as Scenario, et as NoopRawProviderSink, f as ChatCallOpts, h as ChatResponse, i as JudgeFn, it as RawProviderSink, j as LlmMessage, k as LlmClient, l as ProductClientConfig, m as ChatRequest, n as CompletionCriterion, o as JudgeRubric, p as ChatClient, q as JudgeFamily, r as DriverState, rt as RawProviderEvent, s as JudgeScore, t as CheckResult, u as RouteMap, v as CreateChatClientOpts, w as createChatClient, z as costReceiptFromLlmError } from "./types-Cx3YUh2r.js";
|
|
11
11
|
import { a as RunFilter, i as InMemoryTraceStore, n as FileSystemTraceStore, o as SpanFilter, s as TraceStore, t as EventFilter } from "./store-CT9YIIve.js";
|
|
12
12
|
import { i as TraceEmitter, r as SpanHandle } from "./emitter-DGQGoLyj.js";
|
|
13
13
|
import { a as RunIntegrityReport, o as assertRunCaptured, t as RunIntegrityError } from "./integrity-OrcI9Nau.js";
|
|
14
|
-
import { A as createBoundedTraceAnalysisStore, B as OtlpSpan, C as scoreTraceInsightReadiness, D as AnalyzeTracesResult, F as redactString, M as REDACTION_VERSION, O as analyzeTraces, P as RedactionRule, R as OtlpExport, V as exportRunAsOtlp, _ as buildTraceInsightPrompt, b as domainEvidencePattern, g as buildTraceInsightContext, j as DEFAULT_REDACTION_RULES, k as OtlpFlatLine, m as TraceInsightSuite, n as toolSpansToTraceAnalysisStore, o as otlpTextToTraceAnalysisStore, p as TraceInsightReadiness, r as OtlpFileTraceStore, w as tokenizeDomainWords, x as inferDomainKeywords, y as describeTraceInsightScope } from "./store-tool-spans-
|
|
14
|
+
import { A as createBoundedTraceAnalysisStore, B as OtlpSpan, C as scoreTraceInsightReadiness, D as AnalyzeTracesResult, F as redactString, M as REDACTION_VERSION, O as analyzeTraces, P as RedactionRule, R as OtlpExport, V as exportRunAsOtlp, _ as buildTraceInsightPrompt, b as domainEvidencePattern, g as buildTraceInsightContext, j as DEFAULT_REDACTION_RULES, k as OtlpFlatLine, m as TraceInsightSuite, n as toolSpansToTraceAnalysisStore, o as otlpTextToTraceAnalysisStore, p as TraceInsightReadiness, r as OtlpFileTraceStore, w as tokenizeDomainWords, x as inferDomainKeywords, y as describeTraceInsightScope } from "./store-tool-spans-C9c4R7ca.js";
|
|
15
15
|
import { y as OUTPUT_VALUE } from "./attribute-vocabulary-DLJ6303h.js";
|
|
16
16
|
import { a as judgeSpans, c as runsForScenario, n as argHash } from "./query-CJ_DX8vl.js";
|
|
17
|
-
import { A as SearchSpanResult, B as ViewSpansResult, D as DatasetOverview, E as DEFAULT_TRACE_ANALYST_BUDGETS, F as TraceAnalystFilters, H as ViewTraceResult, I as TraceAnalystSpan, L as TraceAnalystSpanKind, M as SpanMatchRecord, N as TRACE_ANALYST_TRUNCATION_MARKER_PREFIX, O as ErrorCluster, P as TraceAnalystByteBudgets, R as TraceAnalystSpanStatus, V as ViewTraceOversized, _ as ProposalFinding, b as makeFinding, c as AnalystRunInputs, d as AnalystSeverity, f as AnalystUsageReceipt, i as AnalystFinding, j as SearchTraceResult, k as QueryTracesPage, l as AnalystRunResult, n as AnalystContext, p as EvidenceRef, s as AnalystRunEvent, t as Analyst, u as AnalystRunSummary, w as TraceAnalysisStore, x as makeProposalFinding, y as computeFindingId, z as TraceAnalystTraceSummary } from "./types-
|
|
18
|
-
import { a as createTraceAnalyst, i as TraceAnalystDefinition, n as buildDefaultAnalystRegistry, r as CreateTraceAnalystOptions, t as DefaultAnalystRegistryOptions } from "./default-registry-
|
|
19
|
-
import { i as ExactAnalystRunEvent, l as ExactCapableAnalyst, o as ExactAnalystRunResult } from "./exact-types-
|
|
20
|
-
import { c as RegistryRunOpts, i as BudgetPolicy, n as AnalystRegistry, r as AnalystRegistryOptions, s as ExactRegistryRunOpts } from "./registry-
|
|
17
|
+
import { A as SearchSpanResult, B as ViewSpansResult, D as DatasetOverview, E as DEFAULT_TRACE_ANALYST_BUDGETS, F as TraceAnalystFilters, H as ViewTraceResult, I as TraceAnalystSpan, L as TraceAnalystSpanKind, M as SpanMatchRecord, N as TRACE_ANALYST_TRUNCATION_MARKER_PREFIX, O as ErrorCluster, P as TraceAnalystByteBudgets, R as TraceAnalystSpanStatus, V as ViewTraceOversized, _ as ProposalFinding, b as makeFinding, c as AnalystRunInputs, d as AnalystSeverity, f as AnalystUsageReceipt, i as AnalystFinding, j as SearchTraceResult, k as QueryTracesPage, l as AnalystRunResult, n as AnalystContext, p as EvidenceRef, s as AnalystRunEvent, t as Analyst, u as AnalystRunSummary, w as TraceAnalysisStore, x as makeProposalFinding, y as computeFindingId, z as TraceAnalystTraceSummary } from "./types-vXyshMwx.js";
|
|
18
|
+
import { a as createTraceAnalyst, i as TraceAnalystDefinition, n as buildDefaultAnalystRegistry, r as CreateTraceAnalystOptions, t as DefaultAnalystRegistryOptions } from "./default-registry-cpHtP2Og.js";
|
|
19
|
+
import { i as ExactAnalystRunEvent, l as ExactCapableAnalyst, o as ExactAnalystRunResult } from "./exact-types-CzbhVDr2.js";
|
|
20
|
+
import { c as RegistryRunOpts, i as BudgetPolicy, n as AnalystRegistry, r as AnalystRegistryOptions, s as ExactRegistryRunOpts } from "./registry-B_1Frl8a.js";
|
|
21
21
|
import { a as ProfileAxisSpec, c as expandProfileAxes, i as HarnessType, l as harnessAxisOf, n as CODING_HARNESSES, o as agentProfileHash, r as HARNESS_NATIVE_MODEL, s as agentProfileId, t as AgentProfile } from "./agent-profile-B9_GGsG8.js";
|
|
22
|
-
import { S as JudgeConfig, a as CampaignResult, w as JudgeScore$1 } from "./types-
|
|
23
|
-
import { _ as SatisfiedBy, a as ArtifactEventLike, b as createLlmCorrectnessChecker, c as ToolCallEventLike, d as CompletionVerdict, f as CorrectnessChecker, g as RequirementCheck, h as ProducedState, i as summarizeBackendIntegrity, l as extractProducedState, m as ProducedProposal, n as BackendIntegrityReport, o as ProposalEventLike, p as LlmCorrectnessCheckerOpts, r as assertRealBackend, s as RuntimeEventLike, t as BackendIntegrityError, u as CompletionRequirement, v as TaskGold, x as verifyCompletion, y as completionVerdict } from "./backend-integrity-
|
|
22
|
+
import { S as JudgeConfig, a as CampaignResult, w as JudgeScore$1 } from "./types-BjsNDR49.js";
|
|
23
|
+
import { _ as SatisfiedBy, a as ArtifactEventLike, b as createLlmCorrectnessChecker, c as ToolCallEventLike, d as CompletionVerdict, f as CorrectnessChecker, g as RequirementCheck, h as ProducedState, i as summarizeBackendIntegrity, l as extractProducedState, m as ProducedProposal, n as BackendIntegrityReport, o as ProposalEventLike, p as LlmCorrectnessCheckerOpts, r as assertRealBackend, s as RuntimeEventLike, t as BackendIntegrityError, u as CompletionRequirement, v as TaskGold, x as verifyCompletion, y as completionVerdict } from "./backend-integrity-BffcHGdm.js";
|
|
24
24
|
import { i as DatasetSplit, n as DatasetManifest, r as DatasetScenario, t as Dataset } from "./dataset-CJjKqQfA.js";
|
|
25
25
|
import { a as ContinuousCalibrationResult, c as calibrateJudge, d as verbosityBias, i as ContinuousAgreementOptions, l as calibrateJudgeContinuous, n as CandidateScore, o as GoldenItem, r as ContinuousAgreement, s as VerbosityBiasResult, t as CalibrationResult, u as continuousAgreement } from "./judge-calibration-C5CbMYce.js";
|
|
26
26
|
import { a as CorpusAgreementPerDimension, c as corpusInterRaterAgreement, i as CorpusAgreementOptions, l as corpusInterRaterAgreementFromJudgeScores, n as SeriesConvergenceResult, o as CorpusAgreementReport, r as analyzeSeries, s as CorpusScoreRecord, t as SeriesConvergenceOptions, u as interRaterReliability } from "./series-convergence-D1cL1f-4.js";
|
|
27
27
|
import { n as bonferroni, r as holm, t as benjaminiHochberg } from "./multiplicity-DIWHvysC.js";
|
|
28
|
-
import { A as RiskDifferenceResult, C as EProcessOptions, D as ExactRiskDifferenceResult, E as eProcess, F as pairedRiskDifference, I as pairedRiskDifferenceExact, L as pairedRiskDifferenceScore, M as isBinaryOutcomeVector, N as mcnemar, O as McNemarResult, P as pairedBinaryScale, R as passAtK, S as EProcess, T as EProcessStep, _ as PairedCorrectness, b as pairArms, d as MatchedRunRecordPair, f as PairArmsOptions, g as PairedArmsComparison, h as PairedArmRow, i as canonicalize, j as ScoreRiskDifferenceResult, k as ProportionInterval, l as ComparePairedArmsOptions, m as PairRunRecordsResult, o as hashJson, p as PairArmsResult, u as MatchedPair, v as PairedMetricDelta, w as EProcessState, x as pairRunRecords, y as comparePairedArms, z as wilson } from "./pre-registration-
|
|
28
|
+
import { A as RiskDifferenceResult, C as EProcessOptions, D as ExactRiskDifferenceResult, E as eProcess, F as pairedRiskDifference, I as pairedRiskDifferenceExact, L as pairedRiskDifferenceScore, M as isBinaryOutcomeVector, N as mcnemar, O as McNemarResult, P as pairedBinaryScale, R as passAtK, S as EProcess, T as EProcessStep, _ as PairedCorrectness, b as pairArms, d as MatchedRunRecordPair, f as PairArmsOptions, g as PairedArmsComparison, h as PairedArmRow, i as canonicalize, j as ScoreRiskDifferenceResult, k as ProportionInterval, l as ComparePairedArmsOptions, m as PairRunRecordsResult, o as hashJson, p as PairArmsResult, u as MatchedPair, v as PairedMetricDelta, w as EProcessState, x as pairRunRecords, y as comparePairedArms, z as wilson } from "./pre-registration-zFSLEiFU.js";
|
|
29
29
|
import { _ as pairedSignTest, a as PairedPromotionDecision, c as BOOTSTRAP_GATE_MIN_N, d as PairedBootstrapResult, f as PairedSignTestResult, g as pairedDeltaTieFraction, h as pairedBootstrap, i as PairedMcNemarEvidence, l as DECISION_PAIRED_DELTA_STATISTIC, m as SignTestAlternative, n as PairedDecisionShape, o as PairedPromotionDecisionOptions, p as PairedTTestResult, r as PairedDecisionStatistic, s as decidePairedPromotion, t as PairedDecisionMethod, u as PairedBootstrapOptions, v as pairedTTest } from "./paired-promotion-decision-CGzg0cI_.js";
|
|
30
30
|
import { _ as requiredSampleSize, c as computeExperimentStats, f as mulberry32, g as requiredPairedSampleSize, h as pairedMde, m as mcnemarRequiredN, n as ExperimentRep, o as ImprovementThresholds, p as mcnemarPower, r as ExperimentStats, s as ImprovementVerdictResult, u as improvementVerdict } from "./experiment-tracker-DWHZBAYL.js";
|
|
31
|
-
import { C as RankTestMethod, D as WilcoxonSignedRankResult, E as WILCOXON_EXACT_MAX_N, O as mannWhitneyU, S as MannWhitneyResult, T as RankTestOptions, _ as bootstrapCi, a as ReleaseConfidenceIssue, b as MANN_WHITNEY_EXACT_MAX_STATES, f as evaluateReleaseConfidence, g as Verdict, i as ReleaseConfidenceInput, k as wilcoxonSignedRank, m as BootstrapResult, o as ReleaseConfidenceMetrics, p as BootstrapOptions, s as ReleaseConfidenceScorecard, t as ActionableSideInfo, u as ReleaseTraceEvidence, w as RankTestMethodRequest, x as MANN_WHITNEY_EXACT_MAX_WORK, y as DEFAULT_PERMUTATIONS } from "./release-confidence-
|
|
32
|
-
import { C as HeldOutGateConfig, S as HeldOutGate, T as SplitCoverage, _ as paretoChart, a as ParetoPoint, b as GateDecision, d as ResearchReportOptions, g as gainHistogram, h as SummaryTableRow, m as SummaryTableOptions, n as GainDistributionFigureSpec, p as SummaryTable, r as GainDistributionOptions, s as ResearchReport, t as GainDistributionBin, w as HeldOutGateRejectionCode, x as GateEvidence, y as summaryTable } from "./summary-report-
|
|
31
|
+
import { C as RankTestMethod, D as WilcoxonSignedRankResult, E as WILCOXON_EXACT_MAX_N, O as mannWhitneyU, S as MannWhitneyResult, T as RankTestOptions, _ as bootstrapCi, a as ReleaseConfidenceIssue, b as MANN_WHITNEY_EXACT_MAX_STATES, f as evaluateReleaseConfidence, g as Verdict, i as ReleaseConfidenceInput, k as wilcoxonSignedRank, m as BootstrapResult, o as ReleaseConfidenceMetrics, p as BootstrapOptions, s as ReleaseConfidenceScorecard, t as ActionableSideInfo, u as ReleaseTraceEvidence, w as RankTestMethodRequest, x as MANN_WHITNEY_EXACT_MAX_WORK, y as DEFAULT_PERMUTATIONS } from "./release-confidence-4XrqlpFD.js";
|
|
32
|
+
import { C as HeldOutGateConfig, S as HeldOutGate, T as SplitCoverage, _ as paretoChart, a as ParetoPoint, b as GateDecision, d as ResearchReportOptions, g as gainHistogram, h as SummaryTableRow, m as SummaryTableOptions, n as GainDistributionFigureSpec, p as SummaryTable, r as GainDistributionOptions, s as ResearchReport, t as GainDistributionBin, w as HeldOutGateRejectionCode, x as GateEvidence, y as summaryTable } from "./summary-report-B__Y5ub3.js";
|
|
33
33
|
import { a as FailureContext, i as FailureClassification, o as FailureRule, r as failureClusterView, s as classifyFailure } from "./failure-cluster-BLURuWG4.js";
|
|
34
|
-
import { o as InsightReport } from "./insight-report-
|
|
35
|
-
import { C as analyzeRuns, b as AnalyzeRunsOptions, d as RawAnalystFinding, i as TraceAnalysisEngineResult, n as TraceAnalysisEngine } from "./engine-
|
|
34
|
+
import { o as InsightReport } from "./insight-report-BeT8KCgI.js";
|
|
35
|
+
import { C as analyzeRuns, b as AnalyzeRunsOptions, d as RawAnalystFinding, i as TraceAnalysisEngineResult, n as TraceAnalysisEngine } from "./engine-DJqRKbhs.js";
|
|
36
36
|
import { o as MintedRolloutLine } from "./schema-Cef2cFmb.js";
|
|
37
|
-
import { i as BenchmarkEvaluation } from "./types-
|
|
38
|
-
import { A as RedTeamCategory, At as llmJudge, B as CanaryReport, F as scoreRedTeamOutput, I as CanaryAlert, L as CanaryEvaluation, M as RedTeamReport, N as redTeamDataset, O as DEFAULT_RED_TEAM_CORPUS, P as redTeamReport, R as CanaryKind, V as runCanaries, at as RunCampaignOptions, it as CampaignCellFailureReceipt, j as RedTeamFinding, k as RedTeamCase, kt as LlmJudgeOptions, ot as runCampaign, z as CanaryOptions } from "./provenance-
|
|
37
|
+
import { i as BenchmarkEvaluation } from "./types-C4bSVIr7.js";
|
|
38
|
+
import { A as RedTeamCategory, At as llmJudge, B as CanaryReport, F as scoreRedTeamOutput, I as CanaryAlert, L as CanaryEvaluation, M as RedTeamReport, N as redTeamDataset, O as DEFAULT_RED_TEAM_CORPUS, P as redTeamReport, R as CanaryKind, V as runCanaries, at as RunCampaignOptions, it as CampaignCellFailureReceipt, j as RedTeamFinding, k as RedTeamCase, kt as LlmJudgeOptions, ot as runCampaign, z as CanaryOptions } from "./provenance-BOtMtoiZ.js";
|
|
39
39
|
import { a as paretoFrontier, i as dominates, n as Objective, r as ParetoResult } from "./pareto-BqNW3LJR.js";
|
|
40
40
|
import { n as SandboxDriver } from "./sandbox-harness-BlSOu4LX.js";
|
|
41
|
-
import { a as DefinedAgentEval, c as SelfImproveOptions, f as selfImprove, i as DefineAgentEvalOptions, o as defineAgentEval, u as SelfImproveResult } from "./define-agent-eval-
|
|
41
|
+
import { a as DefinedAgentEval, c as SelfImproveOptions, f as selfImprove, i as DefineAgentEvalOptions, o as defineAgentEval, u as SelfImproveResult } from "./define-agent-eval-DBOpAM_g.js";
|
|
42
42
|
import { a as PairedEvalueStep, c as pairedEvalueSequence, i as PairedEvalueSequence, n as InterimReleaseConfidenceInput, o as SequentialDecision, r as PairedEvalueOptions, s as evaluateInterimReleaseConfidence, t as InterimReleaseConfidence } from "./sequential-CYwq6Ff_.js";
|
|
43
43
|
import { n as TrajectoryStep, r as buildTrajectory, t as Trajectory } from "./trajectory-YC15QDYQ.js";
|
|
44
44
|
import { a as runCounterfactual, i as CounterfactualRunner, n as CounterfactualMutation, r as CounterfactualResult, t as CounterfactualContext } from "./counterfactual--bpysZF0.js";
|
|
45
|
-
import { A as AnalystFindingDigest, B as ControlRunResult, C as createFeedbackTrajectory, D as renderPreferenceMemoryMarkdown, E as feedbackTrajectoryToOptimizerRow, F as ControlActionOutcome, G as StopDecision, H as ControlRuntimeError, I as ControlBudget, J as subjectiveEval, K as objectiveEval, L as ControlContext, M as AnalystRunDigest, N as analystFindingDigest, O as summarizePreferenceMemory, P as analystRunDigest, R as ControlDecision, S as controlRunToFeedbackTrajectory, T as feedbackTrajectoriesToOptimizerRows, U as ControlSeverity, V as ControlRuntimeConfig, W as ControlStep, _ as PreferenceMemoryEntry, a as FeedbackLabel, b as analystRunToReviewRequests, c as FeedbackOptimizerRow, d as FeedbackTask, f as FeedbackTrajectory, g as InMemoryFeedbackTrajectoryStore, h as FileSystemFeedbackTrajectoryStore, i as FeedbackAttempt, j as AnalystReviewDecision, k as withAssignedFeedbackSplit, l as FeedbackOutcome, m as FeedbackTrajectoryStore, n as AnalystReviewRequest, o as FeedbackLabelKind, p as FeedbackTrajectoryFilter, q as runAgentControlLoop, r as FeedbackArtifactType, s as FeedbackLabelSource, t as AnalystFeedbackTrajectoryOptions, u as FeedbackSplitPolicy, v as ProposedSideEffect, w as feedbackTrajectoriesToDatasetScenarios, x as assignFeedbackSplit, y as analystRunToFeedbackTrajectory, z as ControlEvalResult } from "./feedback-trajectory-
|
|
46
|
-
import { a as SteeringChange, c as CampaignRunContext, d as CampaignVariant, f as EvalCampaignOptions, h as runEvalCampaign, i as Researcher, l as CampaignRunOutcome, m as FailedRun, n as ExperimentResult, o as CampaignFactoryParams, p as EvalCampaignResult, r as FailureMode, s as CampaignIntegrityPolicy, t as ExperimentPlan, u as CampaignScenario } from "./researcher-
|
|
47
|
-
import { J as MintRolloutOptions, Y as MintRolloutResult, Z as mintRolloutRows, et as ScorePreference } from "./index-
|
|
45
|
+
import { A as AnalystFindingDigest, B as ControlRunResult, C as createFeedbackTrajectory, D as renderPreferenceMemoryMarkdown, E as feedbackTrajectoryToOptimizerRow, F as ControlActionOutcome, G as StopDecision, H as ControlRuntimeError, I as ControlBudget, J as subjectiveEval, K as objectiveEval, L as ControlContext, M as AnalystRunDigest, N as analystFindingDigest, O as summarizePreferenceMemory, P as analystRunDigest, R as ControlDecision, S as controlRunToFeedbackTrajectory, T as feedbackTrajectoriesToOptimizerRows, U as ControlSeverity, V as ControlRuntimeConfig, W as ControlStep, _ as PreferenceMemoryEntry, a as FeedbackLabel, b as analystRunToReviewRequests, c as FeedbackOptimizerRow, d as FeedbackTask, f as FeedbackTrajectory, g as InMemoryFeedbackTrajectoryStore, h as FileSystemFeedbackTrajectoryStore, i as FeedbackAttempt, j as AnalystReviewDecision, k as withAssignedFeedbackSplit, l as FeedbackOutcome, m as FeedbackTrajectoryStore, n as AnalystReviewRequest, o as FeedbackLabelKind, p as FeedbackTrajectoryFilter, q as runAgentControlLoop, r as FeedbackArtifactType, s as FeedbackLabelSource, t as AnalystFeedbackTrajectoryOptions, u as FeedbackSplitPolicy, v as ProposedSideEffect, w as feedbackTrajectoriesToDatasetScenarios, x as assignFeedbackSplit, y as analystRunToFeedbackTrajectory, z as ControlEvalResult } from "./feedback-trajectory-D9vYSob_.js";
|
|
46
|
+
import { a as SteeringChange, c as CampaignRunContext, d as CampaignVariant, f as EvalCampaignOptions, h as runEvalCampaign, i as Researcher, l as CampaignRunOutcome, m as FailedRun, n as ExperimentResult, o as CampaignFactoryParams, p as EvalCampaignResult, r as FailureMode, s as CampaignIntegrityPolicy, t as ExperimentPlan, u as CampaignScenario } from "./researcher-Du-oniHp.js";
|
|
47
|
+
import { J as MintRolloutOptions, Y as MintRolloutResult, Z as mintRolloutRows, et as ScorePreference } from "./index-BNPtkBPf.js";
|
|
48
48
|
import { _ as iqr, a as ToolStats, c as computeToolUseMetrics, d as judgeAgreementView, i as toolWasteView, m as budgetBreachView, o as ToolUseMetrics, s as ToolUseOptions } from "./tool-waste-DjRDEsuI.js";
|
|
49
49
|
//#region src/statistics/descriptive.d.ts
|
|
50
50
|
/**
|
|
@@ -2833,5 +2833,5 @@ declare class PairwiseSteeringOptimizer {
|
|
|
2833
2833
|
optimize(rows: SteeringOptimizationRow[], config?: SteeringOptimizerConfig): SteeringOptimizationResult;
|
|
2834
2834
|
}
|
|
2835
2835
|
//#endregion
|
|
2836
|
-
export { AGENT_PROFILE_KINDS, type ActionExecutionPolicy, type ActionPolicyDecision, type ActionableSideInfo, type ActiveLearningOptions, AgentEvalError, type AgentEvalErrorCode, type AgentProfile, type AgentProfileCell, type AgentProfileCellInput, type AgentProfileDimensionValue, type AgentProfileJson, type AgentProfileSourceInput, type Analyst, type AnalystContext, type AnalystFeedbackTrajectoryOptions, type AnalystFinding, type AnalystFindingDigest, AnalystRegistry, type AnalystRegistryOptions, type AnalystReviewDecision, type AnalystReviewRequest, type AnalystRunDigest, type AnalystRunEvent, type AnalystRunInputs, type AnalystRunResult, type AnalystRunSummary, type AnalystSeverity, type AnalystUsageReceipt, type AnalyzeRunsOptions, type AnalyzeTracesResult, type Artifact, type ArtifactEventLike, type AssertCapabilityHeadroomOptions, type AssertCrossFamilyOptions, type AssertSingleBackendOptions, BOOTSTRAP_GATE_MIN_N, type BackendDescriptor, BackendIntegrityError, type BackendIntegrityReport, type BenchmarkEvaluation, type BlendWeights, type BootstrapOptions, type BootstrapResult, BudgetBreachError, BudgetGuard, type BudgetLedgerEntry, type BudgetPolicy, type BudgetSpec, CODING_HARNESSES, type CalibrationResult, type CampaignCellFailureReceipt, type CampaignFactoryParams, type CampaignIntegrityPolicy, type CampaignResult, type CampaignRunContext, type CampaignRunOutcome, type CampaignScenario, type CampaignVariant, type CanaryAlert, type CanaryEvaluation, type CanaryKind, type CanaryLeak, type CanaryOptions, type CanaryReport, type CandidateScore, type CapabilityHeadroomOptions, type CapabilityHeadroomResult, type CellVerdict, type ChannelRollup, type ChatCallOpts, type ChatClient, type ChatRequest, type ChatResponse, type CheckResult, type CheckerIdentity, type CheckerOutcome, type CliffsMagnitude, type CommandRunner, type ComparePairedArmsOptions, type CompletionCriterion, type CompletionRequirement, type CompletionVerdict, type ConceptFinding, type ConceptSpec, ConfigError, type ContinuousAgreement, type ContinuousAgreementOptions, type ContinuousCalibrationResult, type ContractCheckResult, type ContractSpan, type ContractVerdict, type ControlActionOutcome, type ControlBudget, type ControlContext, type ControlDecision, type ControlEvalResult, type ControlRunResult, type ControlRuntimeConfig, type ControlRuntimeError, type ControlStep, type CorpusAgreementOptions, type CorpusAgreementPerDimension, type CorpusAgreementReport, type CorpusScoreRecord, type CorrectnessChecker, CostAccountingIncompleteError, CostCallConflictError, CostCeilingReachedError, type CostChannel, type CostEntry, CostLedger, type CostLedgerFilter, type CostLedgerHandle, type CostLedgerOptions, type CostLedgerPersistence, CostLedgerPersistenceError, type CostLedgerSummary, type CostProvenance, type CostReceipt, CostReceiptCaptureError, type CostReceiptInput, CostReservationExceededError, type CostResult, type CostSummary, CostTracker, type CostUsage, type CounterfactualContext, type CounterfactualMutation, type CounterfactualResult, type CounterfactualRunner, type CreateChatClientOpts, type CreateTraceAnalystOptions, CrossFamilyError, type CustomTokenPricing, DECISION_PAIRED_DELTA_STATISTIC, DEFAULT_PERMUTATIONS, DEFAULT_REDACTION_RULES, DEFAULT_RED_TEAM_CORPUS, DEFAULT_TRACE_ANALYST_BUDGETS, DEFAULT_TRACE_ANALYST_KINDS, type DataAcquisitionPlan, type DatasetManifest, type DatasetOverview, type DatasetScenario, type DatasetSplit, type DecideNextUserTurnOpts, type DefaultAnalystRegistryOptions, type DefaultVerdict, type DefineAgentEvalOptions, type DefinedAgentEval, type DeployGateLayerInput, type DeployRunResult, type DeployRunner, type DetectorEvent, type DetectorSignal, type DiffPolicy, type DiffScorecardOptions, type DirEntry, type DiscoverPersonasOptions, type DiscoveredPersona, type DriverState, type DspyRlmTraceEngineOptions, type EProcess, type EProcessOptions, type EProcessState, type EProcessStep, ERROR_COUNT_PATTERNS, type EnsembleAggregate, type EnsembleJudgeOptions, type EquivalenceArm, type EquivalenceCheckDefinition, type EquivalenceCheckSpec, type EquivalenceChecker, type EquivalenceCheckerInput, type EquivalenceCheckerResult, type EquivalenceObligation, type EquivalenceObligationStatus, EquivalenceProtocolError, type EquivalenceRecord, type ErrorCluster, type ErrorCountPattern, type ErrorStreakOptions, type EvalCampaignOptions, type EvalCampaignResult, type EventFilter, type EvidenceRef, type ExactAnalystRunEvent, type ExactAnalystRunResult, type ExactCapableAnalyst, type ExactRegistryRunOpts, type ExactRiskDifferenceResult, type ExperimentPlan, type ExperimentRep, type ExperimentResult, type ExperimentStats, type ExtractOptions, type ExtractResult, type ExtractedUsage, FAILURE_CLASSES, FAILURE_MODE_KIND_SPEC, type FailedRun, type FailureClass, type FailureClassification, type FailureContext, type FailureMode, type FailureRule, type FeedbackArtifactType, type FeedbackAttempt, type FeedbackLabel, type FeedbackLabelKind, type FeedbackLabelSource, type FeedbackOptimizerRow, type FeedbackOutcome, type FeedbackSplitPolicy, type FeedbackTask, type FeedbackTrajectory, type FeedbackTrajectoryFilter, type FeedbackTrajectoryStore, type FieldDestination, FileSystemFeedbackTrajectoryStore, FileSystemRawProviderSink, FileSystemTraceStore, type Finding, type FindingSubject, type FindingSubjectKind, type FindingsDiff, FindingsStore, type GainDistributionBin, type GainDistributionFigureSpec, type GainDistributionOptions, type GateDecision, type GateEvidence, type GenericSpan, type GoldenItem, HARNESS_NATIVE_MODEL, type HarnessType, type HeadroomInput, HeldOutGate, type HeldOutGateConfig, type HeldOutGateRejectionCode, type HeldOutPartition, type HiddenCriteriaGrader, type HiddenGradeResult, type HiddenLeak, type ImprovementThresholds, type ImprovementVerdictResult, InMemoryFeedbackTrajectoryStore, InMemoryRawProviderSink, InMemoryTraceStore, type InsightReport, type IntentMatchInput, type IntentMatchOptions, type IntentMatchResult, type InterimReleaseConfidence, type InterimReleaseConfidenceInput, JudgeError, type JudgeFamily, type JudgeRetryOutcome, type JudgeRetryPolicy, type JudgeRubric, type JudgeScore, type JudgeScoreInput, type JudgeScoresRecord, type JudgeSpan, type JudgeVerdict, type KeywordConceptSpec, type KeywordCoverageFinding, type KeywordCoverageOptions, type KeywordCoverageResult, type KnowledgeAcquisitionMode, type KnowledgeBundle, type KnowledgeFreshness, type KnowledgeImportance, type KnowledgeReadinessReport, type KnowledgeRequirement, type KnowledgeRequirementCategory, type KnowledgeSensitivity, type Layer, type LayerResult, type LayerStatus, type LeaderboardOptions, type LeaderboardRow, LlmCallError, type LlmCallMetadata, type LlmCallRequest, type LlmCallResult, LlmClient, type LlmClientOptions, type LlmCorrectnessCheckerOpts, type LlmJudgeOptions, type LlmMessage, LlmResponseError, type LlmReviewerConfig, type LlmRouteRequirements, type LlmSpan, MANN_WHITNEY_EXACT_MAX_STATES, MANN_WHITNEY_EXACT_MAX_WORK, MODEL_PRICING, type MannWhitneyResult, type MatchedPair, type MatchedRunRecordPair, type MaximumCharge, type McNemarResult, type MintRolloutOptions, type MintRolloutResult, type MintedRolloutLine, type ModelSeats, MultiLayerVerifier, type NoLeakOptions, NoopRawProviderSink, NotFoundError, OUTPUT_VALUE, type Objective, type Oracle, type OracleObservation, type OracleReport, type OracleResult, type OtlpExport, OtlpFileTraceStore, type OtlpFlatLine, type OtlpSpan, type PaidCallResult, type PairArmsOptions, type PairArmsResult, type PairRunRecordsResult, type PairedArmRow, type PairedArmsComparison, type PairedBootstrapOptions, type PairedBootstrapResult, type PairedCorrectness, type PairedDecisionMethod, type PairedDecisionShape, type PairedDecisionStatistic, type PairedDeltaTestOptions, type PairedDeltaTestResult, type PairedEvalueOptions, type PairedEvalueSequence, type PairedEvalueStep, type PairedMcNemarEvidence, type PairedMetricDelta, type PairedPromotionDecision, type PairedPromotionDecisionOptions, type PairedSignTestResult, type PairedTTestResult, PairwiseSteeringOptimizer, type ParetoPoint, type ParetoResult, type PartitionHeldOutOptions, type PendingCostCall, type PendingCostCallView, type PersistedFinding, type PersonaConfig, type PreferenceMemoryEntry, type ProducedProposal, type ProducedState, type ProductBenchmarkExportOptions, type ProductBenchmarkExportResult, type ProductBenchmarkManifest, type ProductBenchmarkSingleRunExportOptions, type ProductBenchmarkSplit, type ProductBenchmarkValidationReport, ProductClient, type ProductClientConfig, type ProfileAxisSpec, type ProjectRuntimeTrajectoryEvidenceOptions, type PromptHandle, PromptRegistry, type ProportionInterval, type ProposalEventLike, type ProposalFinding, type ProposeFn, type ProposeInput, type ProposeOutput, type ProposeReviewConfig, type ProposeReviewControlAction, type ProposeReviewControlConfig, type ProposeReviewControlResult, type ProposeReviewControlState, type ProposeReviewReport, type ProposedSideEffect, type QueryTracesPage, REDACTION_VERSION, type RankTestMethod, type RankTestMethodRequest, type RankTestOptions, type RawAnalystFinding, type RawProviderEvent, type RawProviderSink, type RecordRunsOptions, type RedTeamCase, type RedTeamCategory, type RedTeamFinding, type RedTeamReport, type RedactionRule, type ReferenceReplayCaseRun, type ReferenceReplayRun, type ReferenceReplaySplit, type ReflectionContext, type ReflectionProposal, type RegistryRunOpts, type ReleaseConfidenceInput, type ReleaseConfidenceIssue, type ReleaseConfidenceMetrics, type ReleaseConfidenceScorecard, type ReleaseTraceEvidence, type RepeatedActionOptions, type RequirementCheck, type ResearchReport, type ResearchReportOptions, type Researcher, type Review, type ReviewFn, type ReviewInput, type ReviewMemoryEntry, type ReviewMemoryStore, type RiskDifferenceResult, type RouteMap, type RoutedField, type Run, type RunCampaignOptions, type RunCommandInput, type RunCommandResult, type RunCostProvenance, type RunFilter, RunIntegrityError, type RunIntegrityReport, type RunJudgeMetadata, type RunOutcome, type RunPaidCallInput, type RunRecord, type RunRecordBackend, RunRecordValidationError, type RunScore, type RunScoreWeights, type RunSplitTag, type RunStatus, type RunTaskFailure, type RunTerminalOutcome, type RunTokenUsage, type RuntimeEventLike, type RuntimeTrajectoryEvidenceProjection, type RuntimeTrajectoryEvidenceSummary, type RuntimeTrajectoryHookEvent, type RuntimeTrajectoryRecord, type RuntimeTrajectoryRunRecord, SEMANTIC_CONCEPT_JUDGE_VERSION, type SandboxDriver, type SatisfiedBy, type Scenario, type ScenarioCost, type ScorePreference, type ScoreRiskDifferenceResult, type Scorecard, type ScorecardCell, type ScorecardCellDiff, type ScorecardDiff, type ScorecardEntry, type ScorecardLogLine, type SearchSpanResult, type SearchTraceResult, type SelfImproveOptions, type SelfImproveResult, type SemanticConceptJudgeInput, type SemanticConceptJudgeOptions, type SemanticConceptJudgeResult, type SequentialDecision, type SeriesConvergenceOptions, type SeriesConvergenceResult, type Severity, type SignTestAlternative, type SingleBackendDivergence, type SingleBackendReport, type Span, type SpanFilter, type SpanHandle, type SpanMatchRecord, type SplitCoverage, type SteeringBundle, type SteeringChange, type SteeringOptimizationResult, type SteeringOptimizationRow, type StopDecision, type StrategyChecker, type StreamingDetector, type SummaryTable, type SummaryTableOptions, type SummaryTableRow, type SynthesisTarget, TRACE_ANALYST_TRUNCATION_MARKER_PREFIX, type TaskGold, type TaskHeadroom, type ToolCallEventLike, type ToolSpan, type ToolStats, type ToolUseMetrics, type ToolUseOptions, type TraceAnalysisEngine, type TraceAnalysisEngineResult, type TraceAnalysisStore, type TraceAnalystByteBudgets, type TraceAnalystDefinition, type TraceAnalystFilters, type TraceAnalystSpan, type TraceAnalystSpanKind, type TraceAnalystSpanStatus, type TraceAnalystTraceSummary, type TraceContract, TraceEmitter, type TraceEvent, type TraceInsightReadiness, type TraceInsightSuite, type TraceStore, type Trajectory, type TrajectoryStep, type TreatmentGate, type TreatmentGateInput, type TreatmentGateOptions, type TrialTrace, type UserQuestion, VERIFICATION_STRATEGIES, VERIFICATION_STRATEGY_SOURCES, ValidationError, type VerbosityBiasResult, type Verdict, type VerdictCacheStore, type VerdictCertification, type Verification, type VerificationReport, type VerificationStrategyProfile, type VerificationStrategySource, type VerifyFn, type VerifyOptions, type ViewSpansResult, type ViewTraceOversized, type ViewTraceResult, type ViteDeployRunnerInput, WILCOXON_EXACT_MAX_N, type WeightedCompositeInput, type WeightedCompositeResult, type WilcoxonSignedRankResult, type WranglerDeployRunnerInput, acquisitionPlansForKnowledgeGaps, agentProfileCellHashMaterial, agentProfileCellKey, agentProfileHash, agentProfileId, aggregateJudgeVerdicts, aggregateRunScore, analystFindingDigest, analystRunDigest, analystRunToFeedbackTrajectory, analystRunToReviewRequests, analyzeAntiSlop, analyzeRuns, analyzeSeries, analyzeTraces, argHash, assertCapabilityHeadroom, assertCrossFamily, assertLlmRoute, assertNoHiddenLeak, assertProductBenchmarkRun, assertRealBackend, assertRunCaptured, assertSingleBackend, assignFeedbackSplit, benjaminiHochberg, blendHeldout, blockingKnowledgeEval, bonferroni, bootstrapCi, budgetBreachView, buildAgentProfileCell, buildDefaultAnalystRegistry, buildEquivalenceRecord, buildReflectionPrompt, buildTraceInsightContext, buildTraceInsightPrompt, buildTrajectory, calibrateJudge, calibrateJudgeContinuous, callLlm, callLlmJson, canonicalJson, canonicalize, capabilityHeadroom, captureFetchToRawSink, certificationEvidenceDigest, checkCanaries, checkTraceContracts, clamp01, classifyFailure, cliffsDelta, cohensD, comparePairedArms, completionVerdict, computeExperimentStats, computeFindingId, computeToolUseMetrics, confidenceInterval, contentHash, continuousAgreement, controlRunToFeedbackTrajectory, corpusInterRaterAgreement, corpusInterRaterAgreementFromJudgeScores, costForTokenPricing, costForUsage, costReceiptFromLlm, costReceiptFromLlmError, createAntiSlopJudge, createBoundedTraceAnalysisStore, createChatClient, createDspyRlmTraceEngine, createFeedbackTrajectory, createLlmCorrectnessChecker, createLlmReviewer, createTraceAnalyst, decideNextUserTurn, decidePairedPromotion, defaultBlendWeights, defineAgentEval, defineEquivalenceCheck, deployGateLayer, describeTraceInsightScope, diffFindings, diffScorecard, discoverPersonas, domainEvidencePattern, dominates, eProcess, ensembleJudge, equivalenceVerdict, errorStreakDetector, estimateCost, estimateTokens, evaluateActionPolicy, evaluateInterimReleaseConfidence, evaluateOracles, evaluateReleaseConfidence, expandProfileAxes, exportProductBenchmark, exportProductBenchmarkRuns, exportRunAsOtlp, extractErrorCount, extractProducedState, extractUsage, extractUsageFromSse, failureClusterView, feedbackTrajectoriesToDatasetScenarios, feedbackTrajectoriesToOptimizerRows, feedbackTrajectoryToOptimizerRow, fileVerdictCache, formatScorecardDiff, gainHistogram, gateTreatmentApplied, gradeOnHidden, gradeSemanticStatus, groupRunsByAgentProfileCell, harnessAxisOf, hashContent, hashJson, hiddenGrade, holm, improvementVerdict, inMemoryReviewStore, inferDomainKeywords, interRaterReliability, interpretCliffs, iqr, isBinaryOutcomeVector, isJudgeSpan, isLlmSpan, isModelPriced, isRunRecord, isToolSpan, isTransientLlmError, jsonShape, jsonlReviewStore, jsonlRunRecordBackend, judgeAgreementView, judgeFamily, judgeSpans, knowledgeReadinessTracePayload, leaderboard, llmJudge, loadScorecard, localCommandRunner, makeFinding, makeProposalFinding, mannWhitneyU, maximumChargeForLlmRequest, mcnemar, mcnemarPower, mcnemarRequiredN, minimumPairsForPairedDeltaTest, mintRolloutRows, modelHasSnapshot, modelPriceKey, mulberry32, notBlocked, objectiveEval, observeAll, otlpTextToTraceAnalysisStore, pairArms, pairRunRecords, pairedBinaryScale, pairedBootstrap, pairedCohensDz, pairedDeltaTest, pairedDeltaTieFraction, pairedEvalueSequence, pairedMde, pairedRiskDifference, pairedRiskDifferenceExact, pairedRiskDifferenceScore, pairedSignTest, pairedTTest, paretoChart, paretoFrontier, parseReflectionResponse, parseRunRecordSafe, parseRuntimeTrajectoryHookEvent, partialCredit, partitionHeldOut, passAtK, pearsonR, preflightModels, probeLlm, productBenchmarkRepoIdentity, index_d_exports as profile, projectRuntimeTrajectoryEvidence, proposeSynthesisTargets, ranks, readProductBenchmarkManifest, recordRuns, recordRunsToScorecard, redTeamDataset, redTeamReport, redactString, regexMatches, renderPreferenceMemoryMarkdown, repeatedActionDetector, requiredPairedSampleSize, requiredSampleSize, resolveModelPricing, resolveSeat, roundTripRunRecord, routeFields, runAgentControlLoop, runCampaign, runCanaries, runCounterfactual, runEquivalenceCheck, runEvalCampaign, runIntentMatchJudge, runKeywordCoverageJudge, runKeywordCoverageJudgeUrl, runProposeReview, runProposeReviewAsControlLoop, runSemanticConceptJudge, runTaskScore, runsForScenario, scoreKnowledgeReadiness, scoreRedTeamOutput, scoreTraceInsightReadiness, seatPresets, selfImprove, spearmanR, stripFencedJson, subjectiveEval, summarizeBackendIntegrity, summarizePreferenceMemory, summaryTable, textInSnapshot, toAgentProfileJson, tokenizeDomainWords, toolSpansToTraceAnalysisStore, toolWasteView, traceContract, urlContains, userQuestionsForKnowledgeGaps, validateRunRecord, verbosityBias, verifyAgentProfileCell, verifyCompletion, viteDeployRunner, weightedComposite, weightedMean, wilcoxonSignedRank, wilson, withAssignedFeedbackSplit, withHeldoutBlend, withJudgeRetry, wranglerDeployRunner };
|
|
2836
|
+
export { AGENT_PROFILE_KINDS, type ActionExecutionPolicy, type ActionPolicyDecision, type ActionableSideInfo, type ActiveLearningOptions, AgentEvalError, type AgentEvalErrorCode, type AgentProfile, type AgentProfileCell, type AgentProfileCellInput, type AgentProfileDimensionValue, type AgentProfileJson, type AgentProfileSourceInput, type Analyst, type AnalystContext, type AnalystFeedbackTrajectoryOptions, type AnalystFinding, type AnalystFindingDigest, AnalystRegistry, type AnalystRegistryOptions, type AnalystReviewDecision, type AnalystReviewRequest, type AnalystRunDigest, type AnalystRunEvent, type AnalystRunInputs, type AnalystRunResult, type AnalystRunSummary, type AnalystSeverity, type AnalystUsageReceipt, type AnalyzeRunsOptions, type AnalyzeTracesResult, type Artifact, type ArtifactEventLike, type AssertCapabilityHeadroomOptions, type AssertCrossFamilyOptions, type AssertSingleBackendOptions, BOOTSTRAP_GATE_MIN_N, type BackendDescriptor, BackendIntegrityError, type BackendIntegrityReport, type BenchmarkEvaluation, type BlendWeights, type BootstrapOptions, type BootstrapResult, BudgetBreachError, BudgetGuard, type BudgetLedgerEntry, type BudgetPolicy, type BudgetSpec, CODING_HARNESSES, type CalibrationResult, type CampaignCellFailureReceipt, type CampaignFactoryParams, type CampaignIntegrityPolicy, type CampaignResult, type CampaignRunContext, type CampaignRunOutcome, type CampaignScenario, type CampaignVariant, type CanaryAlert, type CanaryEvaluation, type CanaryKind, type CanaryLeak, type CanaryOptions, type CanaryReport, type CandidateScore, type CapabilityHeadroomOptions, type CapabilityHeadroomResult, type CellVerdict, type ChannelRollup, type ChatCallOpts, type ChatClient, type ChatRequest, type ChatResponse, type CheckResult, type CheckerIdentity, type CheckerOutcome, type CliffsMagnitude, type CommandRunner, type ComparePairedArmsOptions, type CompletionCriterion, type CompletionRequirement, type CompletionVerdict, type ConceptFinding, type ConceptSpec, ConfigError, type ContinuousAgreement, type ContinuousAgreementOptions, type ContinuousCalibrationResult, type ContractCheckResult, type ContractSpan, type ContractVerdict, type ControlActionOutcome, type ControlBudget, type ControlContext, type ControlDecision, type ControlEvalResult, type ControlRunResult, type ControlRuntimeConfig, type ControlRuntimeError, type ControlStep, type CorpusAgreementOptions, type CorpusAgreementPerDimension, type CorpusAgreementReport, type CorpusScoreRecord, type CorrectnessChecker, CostAccountingIncompleteError, CostCallConflictError, CostCeilingReachedError, type CostChannel, type CostEntry, CostLedger, type CostLedgerFilter, type CostLedgerHandle, type CostLedgerOptions, type CostLedgerPersistence, CostLedgerPersistenceError, type CostLedgerSummary, type CostProvenance, type CostReceipt, CostReceiptCaptureError, type CostReceiptInput, CostReservationExceededError, type CostResult, type CostSummary, CostTracker, type CostUsage, type CounterfactualContext, type CounterfactualMutation, type CounterfactualResult, type CounterfactualRunner, type CreateChatClientOpts, type CreateTraceAnalystOptions, CrossFamilyError, type CustomTokenPricing, DECISION_PAIRED_DELTA_STATISTIC, DEFAULT_PERMUTATIONS, DEFAULT_REDACTION_RULES, DEFAULT_RED_TEAM_CORPUS, DEFAULT_TRACE_ANALYST_BUDGETS, DEFAULT_TRACE_ANALYST_KINDS, type DataAcquisitionPlan, type DatasetManifest, type DatasetOverview, type DatasetScenario, type DatasetSplit, type DecideNextUserTurnOpts, type DefaultAnalystRegistryOptions, type DefaultVerdict, type DefineAgentEvalOptions, type DefinedAgentEval, type DeployGateLayerInput, type DeployRunResult, type DeployRunner, type DetectorEvent, type DetectorSignal, type DiffPolicy, type DiffScorecardOptions, type DirEntry, type DiscoverPersonasOptions, type DiscoveredPersona, type DriverState, type DspyRlmTraceEngineOptions, type EProcess, type EProcessOptions, type EProcessState, type EProcessStep, ERROR_COUNT_PATTERNS, type EnsembleAggregate, type EnsembleJudgeOptions, type EquivalenceArm, type EquivalenceCheckDefinition, type EquivalenceCheckSpec, type EquivalenceChecker, type EquivalenceCheckerInput, type EquivalenceCheckerResult, type EquivalenceObligation, type EquivalenceObligationStatus, EquivalenceProtocolError, type EquivalenceRecord, type ErrorCluster, type ErrorCountPattern, type ErrorStreakOptions, type EvalCampaignOptions, type EvalCampaignResult, type EventFilter, type EvidenceRef, type ExactAnalystRunEvent, type ExactAnalystRunResult, type ExactCapableAnalyst, type ExactRegistryRunOpts, type ExactRiskDifferenceResult, type ExperimentPlan, type ExperimentRep, type ExperimentResult, type ExperimentStats, type ExtractOptions, type ExtractResult, type ExtractedUsage, FAILURE_CLASSES, FAILURE_MODE_KIND_SPEC, type FailedRun, type FailureClass, type FailureClassification, type FailureContext, type FailureMode, type FailureRule, type FeedbackArtifactType, type FeedbackAttempt, type FeedbackLabel, type FeedbackLabelKind, type FeedbackLabelSource, type FeedbackOptimizerRow, type FeedbackOutcome, type FeedbackSplitPolicy, type FeedbackTask, type FeedbackTrajectory, type FeedbackTrajectoryFilter, type FeedbackTrajectoryStore, type FieldDestination, FileSystemFeedbackTrajectoryStore, FileSystemRawProviderSink, FileSystemTraceStore, type Finding, type FindingSubject, type FindingSubjectKind, type FindingsDiff, FindingsStore, type GainDistributionBin, type GainDistributionFigureSpec, type GainDistributionOptions, type GateDecision, type GateEvidence, type GenericSpan, type GoldenItem, HARNESS_NATIVE_MODEL, type HarnessType, type HeadroomInput, HeldOutGate, type HeldOutGateConfig, type HeldOutGateRejectionCode, type HeldOutPartition, type HiddenCriteriaGrader, type HiddenGradeResult, type HiddenLeak, type ImprovementThresholds, type ImprovementVerdictResult, InMemoryFeedbackTrajectoryStore, InMemoryRawProviderSink, InMemoryTraceStore, type InsightReport, type IntentMatchInput, type IntentMatchOptions, type IntentMatchResult, type InterimReleaseConfidence, type InterimReleaseConfidenceInput, JudgeError, type JudgeFamily, type JudgeRetryOutcome, type JudgeRetryPolicy, type JudgeRubric, type JudgeScore, type JudgeScoreInput, type JudgeScoresRecord, type JudgeSpan, type JudgeVerdict, type KeywordConceptSpec, type KeywordCoverageFinding, type KeywordCoverageOptions, type KeywordCoverageResult, type KnowledgeAcquisitionMode, type KnowledgeBundle, type KnowledgeFreshness, type KnowledgeImportance, type KnowledgeReadinessReport, type KnowledgeRequirement, type KnowledgeRequirementCategory, type KnowledgeSensitivity, type Layer, type LayerResult, type LayerStatus, type LeaderboardOptions, type LeaderboardRow, LlmCallError, type LlmCallMetadata, type LlmCallRequest, type LlmCallResult, LlmClient, type LlmClientOptions, type LlmCorrectnessCheckerOpts, type LlmJudgeOptions, type LlmMessage, LlmResponseError, type LlmReviewerConfig, type LlmRouteRequirements, type LlmSpan, MANN_WHITNEY_EXACT_MAX_STATES, MANN_WHITNEY_EXACT_MAX_WORK, MODEL_PRICING, type MannWhitneyResult, type MatchedPair, type MatchedRunRecordPair, type MaximumCharge, type McNemarResult, type MintRolloutOptions, type MintRolloutResult, type MintedRolloutLine, type ModelSeats, MultiLayerVerifier, type NoLeakOptions, NoopRawProviderSink, NotFoundError, OUTPUT_VALUE, type Objective, type Oracle, type OracleObservation, type OracleReport, type OracleResult, type OtlpExport, OtlpFileTraceStore, type OtlpFlatLine, type OtlpSpan, type PaidCallResult, type PairArmsOptions, type PairArmsResult, type PairRunRecordsResult, type PairedArmRow, type PairedArmsComparison, type PairedBootstrapOptions, type PairedBootstrapResult, type PairedCorrectness, type PairedDecisionMethod, type PairedDecisionShape, type PairedDecisionStatistic, type PairedDeltaTestOptions, type PairedDeltaTestResult, type PairedEvalueOptions, type PairedEvalueSequence, type PairedEvalueStep, type PairedMcNemarEvidence, type PairedMetricDelta, type PairedPromotionDecision, type PairedPromotionDecisionOptions, type PairedSignTestResult, type PairedTTestResult, PairwiseSteeringOptimizer, type ParetoPoint, type ParetoResult, type PartitionHeldOutOptions, type PendingCostCall, type PendingCostCallView, type PersistedFinding, type PersonaConfig, type PreferenceMemoryEntry, type ProducedProposal, type ProducedState, type ProductBenchmarkExportOptions, type ProductBenchmarkExportResult, type ProductBenchmarkManifest, type ProductBenchmarkSingleRunExportOptions, type ProductBenchmarkSplit, type ProductBenchmarkValidationReport, ProductClient, type ProductClientConfig, type ProfileAxisSpec, type ProjectRuntimeTrajectoryEvidenceOptions, type PromptHandle, PromptRegistry, type ProportionInterval, type ProposalEventLike, type ProposalFinding, type ProposeFn, type ProposeInput, type ProposeOutput, type ProposeReviewConfig, type ProposeReviewControlAction, type ProposeReviewControlConfig, type ProposeReviewControlResult, type ProposeReviewControlState, type ProposeReviewReport, type ProposedSideEffect, type QueryTracesPage, REDACTION_VERSION, type RankTestMethod, type RankTestMethodRequest, type RankTestOptions, type RawAnalystFinding, type RawProviderEvent, type RawProviderSink, type RecordRunsOptions, type RedTeamCase, type RedTeamCategory, type RedTeamFinding, type RedTeamReport, type RedactionRule, type ReferenceReplayCaseRun, type ReferenceReplayRun, type ReferenceReplaySplit, type ReflectionContext, type ReflectionProposal, type RegistryRunOpts, type ReleaseConfidenceInput, type ReleaseConfidenceIssue, type ReleaseConfidenceMetrics, type ReleaseConfidenceScorecard, type ReleaseTraceEvidence, type RepeatedActionOptions, type RequirementCheck, type ResearchReport, type ResearchReportOptions, type Researcher, type Review, type ReviewFn, type ReviewInput, type ReviewMemoryEntry, type ReviewMemoryStore, type RiskDifferenceResult, type RouteMap, type RoutedField, type Run, type RunCampaignOptions, type RunCommandInput, type RunCommandResult, type RunCostProvenance, type RunFilter, RunIntegrityError, type RunIntegrityReport, type RunJudgeMetadata, type RunOutcome, type RunPaidCallInput, type RunRecord, type RunRecordBackend, RunRecordValidationError, type RunScore, type RunScoreWeights, type RunSplitTag, type RunStatus, type RunTaskFailure, type RunTerminalOutcome, type RunTokenUsage, type RuntimeEventLike, type RuntimeTrajectoryEvidenceProjection, type RuntimeTrajectoryEvidenceSummary, type RuntimeTrajectoryHookEvent, type RuntimeTrajectoryRecord, type RuntimeTrajectoryRunRecord, SEMANTIC_CONCEPT_JUDGE_VERSION, type SandboxDriver, type SatisfiedBy, type Scenario, type ScenarioCost, type ScorePreference, type ScoreRiskDifferenceResult, type Scorecard, type ScorecardCell, type ScorecardCellDiff, type ScorecardDiff, type ScorecardEntry, type ScorecardLogLine, type SearchSpanResult, type SearchTraceResult, type SelfImproveOptions, type SelfImproveResult, type SemanticConceptJudgeInput, type SemanticConceptJudgeOptions, type SemanticConceptJudgeResult, type SequentialDecision, type SeriesConvergenceOptions, type SeriesConvergenceResult, type Severity, type SignTestAlternative, type SingleBackendDivergence, type SingleBackendReport, type Span, type SpanFilter, type SpanHandle, type SpanMatchRecord, type SplitCoverage, type SteeringBundle, type SteeringChange, type SteeringOptimizationResult, type SteeringOptimizationRow, type StopDecision, type StrategyChecker, type StreamingDetector, type SummaryTable, type SummaryTableOptions, type SummaryTableRow, type SynthesisTarget, TRACE_ANALYST_TRUNCATION_MARKER_PREFIX, type TaskGold, type TaskHeadroom, type ToolCallEventLike, type ToolSpan, type ToolStats, type ToolUseMetrics, type ToolUseOptions, type TraceAnalysisEngine, type TraceAnalysisEngineResult, type TraceAnalysisStore, type TraceAnalystByteBudgets, type TraceAnalystDefinition, type TraceAnalystFilters, type TraceAnalystSpan, type TraceAnalystSpanKind, type TraceAnalystSpanStatus, type TraceAnalystTraceSummary, type TraceContract, TraceEmitter, type TraceEvent, type TraceInsightReadiness, type TraceInsightSuite, type TraceStore, type Trajectory, type TrajectoryStep, type TreatmentGate, type TreatmentGateInput, type TreatmentGateOptions, type TrialTrace, UNKNOWN_MODEL, type UserQuestion, VERIFICATION_STRATEGIES, VERIFICATION_STRATEGY_SOURCES, ValidationError, type VerbosityBiasResult, type Verdict, type VerdictCacheStore, type VerdictCertification, type Verification, type VerificationReport, type VerificationStrategyProfile, type VerificationStrategySource, type VerifyFn, type VerifyOptions, type ViewSpansResult, type ViewTraceOversized, type ViewTraceResult, type ViteDeployRunnerInput, WILCOXON_EXACT_MAX_N, type WeightedCompositeInput, type WeightedCompositeResult, type WilcoxonSignedRankResult, type WranglerDeployRunnerInput, acquisitionPlansForKnowledgeGaps, agentProfileCellHashMaterial, agentProfileCellKey, agentProfileHash, agentProfileId, aggregateJudgeVerdicts, aggregateRunScore, analystFindingDigest, analystRunDigest, analystRunToFeedbackTrajectory, analystRunToReviewRequests, analyzeAntiSlop, analyzeRuns, analyzeSeries, analyzeTraces, argHash, assertCapabilityHeadroom, assertCrossFamily, assertLlmRoute, assertNoHiddenLeak, assertProductBenchmarkRun, assertRealBackend, assertRunCaptured, assertSingleBackend, assignFeedbackSplit, benjaminiHochberg, blendHeldout, blockingKnowledgeEval, bonferroni, bootstrapCi, budgetBreachView, buildAgentProfileCell, buildDefaultAnalystRegistry, buildEquivalenceRecord, buildReflectionPrompt, buildTraceInsightContext, buildTraceInsightPrompt, buildTrajectory, calibrateJudge, calibrateJudgeContinuous, callLlm, callLlmJson, canonicalJson, canonicalize, capabilityHeadroom, captureFetchToRawSink, certificationEvidenceDigest, checkCanaries, checkTraceContracts, clamp01, classifyFailure, cliffsDelta, cohensD, comparePairedArms, completionVerdict, computeExperimentStats, computeFindingId, computeToolUseMetrics, confidenceInterval, contentHash, continuousAgreement, controlRunToFeedbackTrajectory, corpusInterRaterAgreement, corpusInterRaterAgreementFromJudgeScores, costForTokenPricing, costForUsage, costReceiptFromLlm, costReceiptFromLlmError, createAntiSlopJudge, createBoundedTraceAnalysisStore, createChatClient, createDspyRlmTraceEngine, createFeedbackTrajectory, createLlmCorrectnessChecker, createLlmReviewer, createTraceAnalyst, decideNextUserTurn, decidePairedPromotion, defaultBlendWeights, defineAgentEval, defineEquivalenceCheck, deployGateLayer, describeTraceInsightScope, diffFindings, diffScorecard, discoverPersonas, domainEvidencePattern, dominates, eProcess, ensembleJudge, equivalenceVerdict, errorStreakDetector, estimateCost, estimateTokens, evaluateActionPolicy, evaluateInterimReleaseConfidence, evaluateOracles, evaluateReleaseConfidence, expandProfileAxes, exportProductBenchmark, exportProductBenchmarkRuns, exportRunAsOtlp, extractErrorCount, extractProducedState, extractUsage, extractUsageFromSse, failureClusterView, feedbackTrajectoriesToDatasetScenarios, feedbackTrajectoriesToOptimizerRows, feedbackTrajectoryToOptimizerRow, fileVerdictCache, formatScorecardDiff, gainHistogram, gateTreatmentApplied, gradeOnHidden, gradeSemanticStatus, groupRunsByAgentProfileCell, harnessAxisOf, hashContent, hashJson, hiddenGrade, holm, improvementVerdict, inMemoryReviewStore, inferDomainKeywords, interRaterReliability, interpretCliffs, iqr, isBinaryOutcomeVector, isJudgeSpan, isLlmSpan, isModelPriced, isRunRecord, isToolSpan, isTransientLlmError, jsonShape, jsonlReviewStore, jsonlRunRecordBackend, judgeAgreementView, judgeFamily, judgeSpans, knowledgeReadinessTracePayload, leaderboard, llmJudge, loadScorecard, localCommandRunner, makeFinding, makeProposalFinding, mannWhitneyU, maximumChargeForLlmRequest, mcnemar, mcnemarPower, mcnemarRequiredN, minimumPairsForPairedDeltaTest, mintRolloutRows, modelHasSnapshot, modelPriceKey, mulberry32, notBlocked, objectiveEval, observeAll, otlpTextToTraceAnalysisStore, pairArms, pairRunRecords, pairedBinaryScale, pairedBootstrap, pairedCohensDz, pairedDeltaTest, pairedDeltaTieFraction, pairedEvalueSequence, pairedMde, pairedRiskDifference, pairedRiskDifferenceExact, pairedRiskDifferenceScore, pairedSignTest, pairedTTest, paretoChart, paretoFrontier, parseReflectionResponse, parseRunRecordSafe, parseRuntimeTrajectoryHookEvent, partialCredit, partitionHeldOut, passAtK, pearsonR, preflightModels, probeLlm, productBenchmarkRepoIdentity, index_d_exports as profile, projectRuntimeTrajectoryEvidence, proposeSynthesisTargets, ranks, readProductBenchmarkManifest, recordRuns, recordRunsToScorecard, redTeamDataset, redTeamReport, redactString, regexMatches, renderPreferenceMemoryMarkdown, repeatedActionDetector, requiredPairedSampleSize, requiredSampleSize, resolveModelPricing, resolveSeat, roundTripRunRecord, routeFields, runAgentControlLoop, runCampaign, runCanaries, runCounterfactual, runEquivalenceCheck, runEvalCampaign, runIntentMatchJudge, runKeywordCoverageJudge, runKeywordCoverageJudgeUrl, runProposeReview, runProposeReviewAsControlLoop, runSemanticConceptJudge, runTaskScore, runsForScenario, scoreKnowledgeReadiness, scoreRedTeamOutput, scoreTraceInsightReadiness, seatPresets, selfImprove, spearmanR, stripFencedJson, subjectiveEval, summarizeBackendIntegrity, summarizePreferenceMemory, summaryTable, textInSnapshot, toAgentProfileJson, tokenizeDomainWords, toolSpansToTraceAnalysisStore, toolWasteView, traceContract, urlContains, userQuestionsForKnowledgeGaps, validateRunRecord, verbosityBias, verifyAgentProfileCell, verifyCompletion, viteDeployRunner, weightedComposite, weightedMean, wilcoxonSignedRank, wilson, withAssignedFeedbackSplit, withHeldoutBlend, withJudgeRetry, wranglerDeployRunner };
|
|
2837
2837
|
//# sourceMappingURL=index.d.ts.map
|
package/dist/index.js
CHANGED
|
@@ -1,7 +1,7 @@
|
|
|
1
1
|
import { t as __exportAll } from "./rolldown-runtime-8H4AJuhK.js";
|
|
2
2
|
import { i as JudgeError, o as NotFoundError, r as ConfigError, s as ValidationError, t as AgentEvalError } from "./errors-Dngq5h35.js";
|
|
3
3
|
import { r as hashJson, t as canonicalize } from "./pre-registration-DakwTRXk.js";
|
|
4
|
-
import { a as CODING_HARNESSES, c as agentProfileId, d as harnessAxisOf, i as verifyCompletion, l as agentProfileModelId, n as completionVerdict, o as HARNESS_NATIVE_MODEL, r as createLlmCorrectnessChecker, s as agentProfileHash, t as extractProducedState, u as expandProfileAxes } from "./produced-state-
|
|
4
|
+
import { a as CODING_HARNESSES, c as agentProfileId, d as harnessAxisOf, i as verifyCompletion, l as agentProfileModelId, n as completionVerdict, o as HARNESS_NATIVE_MODEL, r as createLlmCorrectnessChecker, s as agentProfileHash, t as extractProducedState, u as expandProfileAxes } from "./produced-state-C6vSgOpE.js";
|
|
5
5
|
import { AGENT_PROFILE_KINDS, agentProfileCellHashMaterial, agentProfileCellKey, buildAgentProfileCell, groupRunsByAgentProfileCell, toAgentProfileJson, verifyAgentProfileCell } from "./profile-cell.js";
|
|
6
6
|
import { c as mulberry32 } from "./internal-BDHPCnjk.js";
|
|
7
7
|
import { a as spearmanR, i as ranks, n as partialCredit, o as weightedComposite, r as pearsonR, s as weightedMean, t as confidenceInterval } from "./descriptive-B5MwKfbf.js";
|
|
@@ -15,12 +15,12 @@ import { t as eProcess } from "./sequential-eprocess-CbUt2htw.js";
|
|
|
15
15
|
import { n as iqr, r as welchsTTest } from "./baseline-BhPRQBVn.js";
|
|
16
16
|
import { i as isLlmSpan, r as isJudgeSpan, s as isToolSpan, t as FAILURE_CLASSES } from "./schema-CRhEY1SO.js";
|
|
17
17
|
import { a as judgeSpans, c as runsForScenario, n as argHash } from "./query-Di7eEQ79.js";
|
|
18
|
-
import { i as analyzeRuns, o as checkCanaries, r as selfImprove, t as defineAgentEval } from "./define-agent-eval-
|
|
18
|
+
import { i as analyzeRuns, o as checkCanaries, r as selfImprove, t as defineAgentEval } from "./define-agent-eval-m5lM7XGE.js";
|
|
19
19
|
import { r as observedSplitScore, t as isRealnessGated } from "./reward-nw2xZGZG.js";
|
|
20
|
-
import { a as
|
|
20
|
+
import { a as parseRunRecordSafe, c as validateRunRecord, i as modelHasSnapshot, n as UNKNOWN_MODEL, o as roundTripRunRecord, r as isRunRecord, s as runTaskScore, t as RunRecordValidationError } from "./run-record-D2lDdSAz.js";
|
|
21
21
|
import { a as summaryTable, n as gainHistogram, r as paretoChart } from "./summary-report-Blysd6Z2.js";
|
|
22
22
|
import { n as contentHash, r as fileVerdictCache, t as canonicalJson } from "./verdict-cache-mZf5FEiY.js";
|
|
23
|
-
import { B as DEFAULT_RED_TEAM_CORPUS, F as parseReflectionResponse, H as redTeamReport, K as runCampaign, P as buildReflectionPrompt, Q as summarizeBackendIntegrity, U as scoreRedTeamOutput, V as redTeamDataset, W as runCanaries, X as BackendIntegrityError, Z as assertRealBackend, _ as paretoFrontier, g as dominates, k as surfaceContentHash, t as llmJudge } from "./llm-judge-
|
|
23
|
+
import { B as DEFAULT_RED_TEAM_CORPUS, F as parseReflectionResponse, H as redTeamReport, K as runCampaign, P as buildReflectionPrompt, Q as summarizeBackendIntegrity, U as scoreRedTeamOutput, V as redTeamDataset, W as runCanaries, X as BackendIntegrityError, Z as assertRealBackend, _ as paretoFrontier, g as dominates, k as surfaceContentHash, t as llmJudge } from "./llm-judge-BO1LGjdn.js";
|
|
24
24
|
import { a as resolveModelPricing, i as isModelPriced, n as estimateCost, r as estimateTokens, t as MODEL_PRICING } from "./metrics-Cl0L1KUy.js";
|
|
25
25
|
import { a as CostLedgerPersistenceError, c as costForTokenPricing, i as CostLedger, l as costForUsage, n as CostCallConflictError, o as CostReceiptCaptureError, r as CostCeilingReachedError, s as CostReservationExceededError, t as CostAccountingIncompleteError, u as modelPriceKey } from "./cost-ledger-BSe92yAV.js";
|
|
26
26
|
import { n as REDACTION_VERSION, r as redactString, t as DEFAULT_REDACTION_RULES } from "./redact-7Aq1ukl-.js";
|
|
@@ -41,16 +41,16 @@ import { n as assertRunCaptured, t as RunIntegrityError } from "./integrity-Cy9W
|
|
|
41
41
|
import { n as InMemoryTraceStore, t as FileSystemTraceStore } from "./store-DNe_Uv1Q.js";
|
|
42
42
|
import { t as packageVersion$1 } from "./package-version-D7lQHt_-.js";
|
|
43
43
|
import { _ as checkServedModel, a as assertLlmRoute, b as judgeFamily, c as callLlmJson, d as isTransientLlmError, f as maximumChargeForLlmRequest, g as assertServedModel, l as costReceiptFromLlm, m as stripFencedJson, n as LlmClient, o as backoffMs, p as probeLlm, r as LlmResponseError, s as callLlm, t as LlmCallError, u as costReceiptFromLlmError, v as CrossFamilyError, y as assertCrossFamily } from "./llm-client-d0-2TT1g.js";
|
|
44
|
-
import { t as runEvalCampaign } from "./eval-campaign-
|
|
44
|
+
import { t as runEvalCampaign } from "./eval-campaign-aNqpefCS.js";
|
|
45
45
|
import { i as improvementVerdict, n as computeExperimentStats } from "./experiment-tracker-C29gXM4B.js";
|
|
46
|
-
import "./rollout-
|
|
47
|
-
import { t as mintRolloutRows } from "./mint-
|
|
46
|
+
import "./rollout-ytVQ7WT8.js";
|
|
47
|
+
import { t as mintRolloutRows } from "./mint-BV6tLVWl.js";
|
|
48
48
|
import { n as pairedEvalueSequence, t as evaluateInterimReleaseConfidence } from "./sequential-Br0mAPHA.js";
|
|
49
49
|
import { t as createDspyRlmTraceEngine } from "./dspy-rlm-engine-BZ2aMZEd.js";
|
|
50
50
|
import { a as diffFindings, n as runSemanticConceptJudge, r as FindingsStore, t as SEMANTIC_CONCEPT_JUDGE_VERSION } from "./semantic-concept-judge-BiJxScqe.js";
|
|
51
51
|
import { t as analyzeSeries } from "./series-convergence-CjO2QdRW.js";
|
|
52
52
|
import { i as otlpTextToTraceAnalysisStore, n as OtlpFileTraceStore } from "./store-otlp-CsptLYpN.js";
|
|
53
|
-
import { n as evaluateReleaseConfidence, r as bootstrapCi } from "./release-confidence-
|
|
53
|
+
import { n as evaluateReleaseConfidence, r as bootstrapCi } from "./release-confidence-BknrpBnO.js";
|
|
54
54
|
import { createHash } from "node:crypto";
|
|
55
55
|
import { accessSync, appendFileSync, constants, cpSync, existsSync, mkdirSync, promises, readFileSync, readdirSync, statSync, writeFileSync } from "node:fs";
|
|
56
56
|
import { basename, delimiter, dirname, extname, isAbsolute, join, relative, resolve } from "node:path";
|
|
@@ -7402,6 +7402,6 @@ function rankRows(rows, weights) {
|
|
|
7402
7402
|
})).sort((a, b) => b.mean - a.mean);
|
|
7403
7403
|
}
|
|
7404
7404
|
//#endregion
|
|
7405
|
-
export { AGENT_PROFILE_KINDS, AgentEvalError, AnalystRegistry, BOOTSTRAP_GATE_MIN_N, BackendIntegrityError, BudgetBreachError, BudgetGuard, CODING_HARNESSES, ConfigError, CostAccountingIncompleteError, CostCallConflictError, CostCeilingReachedError, CostLedger, CostLedgerPersistenceError, CostReceiptCaptureError, CostReservationExceededError, CostTracker, CrossFamilyError, DECISION_PAIRED_DELTA_STATISTIC, DEFAULT_PERMUTATIONS, DEFAULT_REDACTION_RULES, DEFAULT_RED_TEAM_CORPUS, DEFAULT_TRACE_ANALYST_BUDGETS, DEFAULT_TRACE_ANALYST_KINDS, ERROR_COUNT_PATTERNS, EquivalenceProtocolError, FAILURE_CLASSES, FAILURE_MODE_KIND_SPEC, FileSystemFeedbackTrajectoryStore, FileSystemRawProviderSink, FileSystemTraceStore, FindingsStore, HARNESS_NATIVE_MODEL, HeldOutGate, InMemoryFeedbackTrajectoryStore, InMemoryRawProviderSink, InMemoryTraceStore, JudgeError, LlmCallError, LlmClient, LlmResponseError, MANN_WHITNEY_EXACT_MAX_STATES, MANN_WHITNEY_EXACT_MAX_WORK, MODEL_PRICING, MultiLayerVerifier, NoopRawProviderSink, NotFoundError, OUTPUT_VALUE, OtlpFileTraceStore, PairwiseSteeringOptimizer, ProductClient, PromptRegistry, REDACTION_VERSION, RunIntegrityError, RunRecordValidationError, SEMANTIC_CONCEPT_JUDGE_VERSION, TRACE_ANALYST_TRUNCATION_MARKER_PREFIX, TraceEmitter, VERIFICATION_STRATEGIES, VERIFICATION_STRATEGY_SOURCES, ValidationError, WILCOXON_EXACT_MAX_N, acquisitionPlansForKnowledgeGaps, agentProfileCellHashMaterial, agentProfileCellKey, agentProfileHash, agentProfileId, aggregateJudgeVerdicts, aggregateRunScore, analystFindingDigest, analystRunDigest, analystRunToFeedbackTrajectory, analystRunToReviewRequests, analyzeAntiSlop, analyzeRuns, analyzeSeries, analyzeTraces, argHash, assertCapabilityHeadroom, assertCrossFamily, assertLlmRoute, assertNoHiddenLeak, assertProductBenchmarkRun, assertRealBackend, assertRunCaptured, assertSingleBackend, assignFeedbackSplit, benjaminiHochberg, blendHeldout, blockingKnowledgeEval, bonferroni, bootstrapCi, budgetBreachView, buildAgentProfileCell, buildDefaultAnalystRegistry, buildEquivalenceRecord, buildReflectionPrompt, buildTraceInsightContext, buildTraceInsightPrompt, buildTrajectory, calibrateJudge, calibrateJudgeContinuous, callLlm, callLlmJson, canonicalJson, canonicalize, capabilityHeadroom, captureFetchToRawSink, certificationEvidenceDigest, checkCanaries, checkTraceContracts, clamp01, classifyFailure, cliffsDelta, cohensD, comparePairedArms, completionVerdict, computeExperimentStats, computeFindingId, computeToolUseMetrics, confidenceInterval, contentHash, continuousAgreement, controlRunToFeedbackTrajectory, corpusInterRaterAgreement, corpusInterRaterAgreementFromJudgeScores, costForTokenPricing, costForUsage, costReceiptFromLlm, costReceiptFromLlmError, createAntiSlopJudge, createBoundedTraceAnalysisStore, createChatClient, createDspyRlmTraceEngine, createFeedbackTrajectory, createLlmCorrectnessChecker, createLlmReviewer, createTraceAnalyst, decideNextUserTurn, decidePairedPromotion, defaultBlendWeights, defineAgentEval, defineEquivalenceCheck, deployGateLayer, describeTraceInsightScope, diffFindings, diffScorecard, discoverPersonas, domainEvidencePattern, dominates, eProcess, ensembleJudge, equivalenceVerdict, errorStreakDetector, estimateCost, estimateTokens, evaluateActionPolicy, evaluateInterimReleaseConfidence, evaluateOracles, evaluateReleaseConfidence, expandProfileAxes, exportProductBenchmark, exportProductBenchmarkRuns, exportRunAsOtlp, extractErrorCount, extractProducedState, extractUsage, extractUsageFromSse, failureClusterView, feedbackTrajectoriesToDatasetScenarios, feedbackTrajectoriesToOptimizerRows, feedbackTrajectoryToOptimizerRow, fileVerdictCache, formatScorecardDiff, gainHistogram, gateTreatmentApplied, gradeOnHidden, gradeSemanticStatus, groupRunsByAgentProfileCell, harnessAxisOf, hashContent, hashJson, hiddenGrade, holm, improvementVerdict, inMemoryReviewStore, inferDomainKeywords, interRaterReliability, interpretCliffs, iqr, isBinaryOutcomeVector, isJudgeSpan, isLlmSpan, isModelPriced, isRunRecord, isToolSpan, isTransientLlmError, jsonShape, jsonlReviewStore, jsonlRunRecordBackend, judgeAgreementView, judgeFamily, judgeSpans, knowledgeReadinessTracePayload, leaderboard, llmJudge, loadScorecard, localCommandRunner, makeFinding, makeProposalFinding, mannWhitneyU, maximumChargeForLlmRequest, mcnemar, mcnemarPower, mcnemarRequiredN, minimumPairsForPairedDeltaTest, mintRolloutRows, modelHasSnapshot, modelPriceKey, mulberry32, notBlocked, objectiveEval, observeAll, otlpTextToTraceAnalysisStore, pairArms, pairRunRecords, pairedBinaryScale, pairedBootstrap, pairedCohensDz, pairedDeltaTest, pairedDeltaTieFraction, pairedEvalueSequence, pairedMde, pairedRiskDifference, pairedRiskDifferenceExact, pairedRiskDifferenceScore, pairedSignTest, pairedTTest, paretoChart, paretoFrontier, parseReflectionResponse, parseRunRecordSafe, parseRuntimeTrajectoryHookEvent, partialCredit, partitionHeldOut, passAtK, pearsonR, preflightModels, probeLlm, productBenchmarkRepoIdentity, profile_exports as profile, projectRuntimeTrajectoryEvidence, proposeSynthesisTargets, ranks, readProductBenchmarkManifest, recordRuns, recordRunsToScorecard, redTeamDataset, redTeamReport, redactString, regexMatches, renderPreferenceMemoryMarkdown, repeatedActionDetector, requiredPairedSampleSize, requiredSampleSize, resolveModelPricing, resolveSeat, roundTripRunRecord, routeFields, runAgentControlLoop, runCampaign, runCanaries, runCounterfactual, runEquivalenceCheck, runEvalCampaign, runIntentMatchJudge, runKeywordCoverageJudge, runKeywordCoverageJudgeUrl, runProposeReview, runProposeReviewAsControlLoop, runSemanticConceptJudge, runTaskScore, runsForScenario, scoreKnowledgeReadiness, scoreRedTeamOutput, scoreTraceInsightReadiness, seatPresets, selfImprove, spearmanR, stripFencedJson, subjectiveEval, summarizeBackendIntegrity, summarizePreferenceMemory, summaryTable, textInSnapshot, toAgentProfileJson, tokenizeDomainWords, toolSpansToTraceAnalysisStore, toolWasteView, traceContract, urlContains, userQuestionsForKnowledgeGaps, validateRunRecord, verbosityBias, verifyAgentProfileCell, verifyCompletion, viteDeployRunner, weightedComposite, weightedMean, wilcoxonSignedRank, wilson, withAssignedFeedbackSplit, withHeldoutBlend, withJudgeRetry, wranglerDeployRunner };
|
|
7405
|
+
export { AGENT_PROFILE_KINDS, AgentEvalError, AnalystRegistry, BOOTSTRAP_GATE_MIN_N, BackendIntegrityError, BudgetBreachError, BudgetGuard, CODING_HARNESSES, ConfigError, CostAccountingIncompleteError, CostCallConflictError, CostCeilingReachedError, CostLedger, CostLedgerPersistenceError, CostReceiptCaptureError, CostReservationExceededError, CostTracker, CrossFamilyError, DECISION_PAIRED_DELTA_STATISTIC, DEFAULT_PERMUTATIONS, DEFAULT_REDACTION_RULES, DEFAULT_RED_TEAM_CORPUS, DEFAULT_TRACE_ANALYST_BUDGETS, DEFAULT_TRACE_ANALYST_KINDS, ERROR_COUNT_PATTERNS, EquivalenceProtocolError, FAILURE_CLASSES, FAILURE_MODE_KIND_SPEC, FileSystemFeedbackTrajectoryStore, FileSystemRawProviderSink, FileSystemTraceStore, FindingsStore, HARNESS_NATIVE_MODEL, HeldOutGate, InMemoryFeedbackTrajectoryStore, InMemoryRawProviderSink, InMemoryTraceStore, JudgeError, LlmCallError, LlmClient, LlmResponseError, MANN_WHITNEY_EXACT_MAX_STATES, MANN_WHITNEY_EXACT_MAX_WORK, MODEL_PRICING, MultiLayerVerifier, NoopRawProviderSink, NotFoundError, OUTPUT_VALUE, OtlpFileTraceStore, PairwiseSteeringOptimizer, ProductClient, PromptRegistry, REDACTION_VERSION, RunIntegrityError, RunRecordValidationError, SEMANTIC_CONCEPT_JUDGE_VERSION, TRACE_ANALYST_TRUNCATION_MARKER_PREFIX, TraceEmitter, UNKNOWN_MODEL, VERIFICATION_STRATEGIES, VERIFICATION_STRATEGY_SOURCES, ValidationError, WILCOXON_EXACT_MAX_N, acquisitionPlansForKnowledgeGaps, agentProfileCellHashMaterial, agentProfileCellKey, agentProfileHash, agentProfileId, aggregateJudgeVerdicts, aggregateRunScore, analystFindingDigest, analystRunDigest, analystRunToFeedbackTrajectory, analystRunToReviewRequests, analyzeAntiSlop, analyzeRuns, analyzeSeries, analyzeTraces, argHash, assertCapabilityHeadroom, assertCrossFamily, assertLlmRoute, assertNoHiddenLeak, assertProductBenchmarkRun, assertRealBackend, assertRunCaptured, assertSingleBackend, assignFeedbackSplit, benjaminiHochberg, blendHeldout, blockingKnowledgeEval, bonferroni, bootstrapCi, budgetBreachView, buildAgentProfileCell, buildDefaultAnalystRegistry, buildEquivalenceRecord, buildReflectionPrompt, buildTraceInsightContext, buildTraceInsightPrompt, buildTrajectory, calibrateJudge, calibrateJudgeContinuous, callLlm, callLlmJson, canonicalJson, canonicalize, capabilityHeadroom, captureFetchToRawSink, certificationEvidenceDigest, checkCanaries, checkTraceContracts, clamp01, classifyFailure, cliffsDelta, cohensD, comparePairedArms, completionVerdict, computeExperimentStats, computeFindingId, computeToolUseMetrics, confidenceInterval, contentHash, continuousAgreement, controlRunToFeedbackTrajectory, corpusInterRaterAgreement, corpusInterRaterAgreementFromJudgeScores, costForTokenPricing, costForUsage, costReceiptFromLlm, costReceiptFromLlmError, createAntiSlopJudge, createBoundedTraceAnalysisStore, createChatClient, createDspyRlmTraceEngine, createFeedbackTrajectory, createLlmCorrectnessChecker, createLlmReviewer, createTraceAnalyst, decideNextUserTurn, decidePairedPromotion, defaultBlendWeights, defineAgentEval, defineEquivalenceCheck, deployGateLayer, describeTraceInsightScope, diffFindings, diffScorecard, discoverPersonas, domainEvidencePattern, dominates, eProcess, ensembleJudge, equivalenceVerdict, errorStreakDetector, estimateCost, estimateTokens, evaluateActionPolicy, evaluateInterimReleaseConfidence, evaluateOracles, evaluateReleaseConfidence, expandProfileAxes, exportProductBenchmark, exportProductBenchmarkRuns, exportRunAsOtlp, extractErrorCount, extractProducedState, extractUsage, extractUsageFromSse, failureClusterView, feedbackTrajectoriesToDatasetScenarios, feedbackTrajectoriesToOptimizerRows, feedbackTrajectoryToOptimizerRow, fileVerdictCache, formatScorecardDiff, gainHistogram, gateTreatmentApplied, gradeOnHidden, gradeSemanticStatus, groupRunsByAgentProfileCell, harnessAxisOf, hashContent, hashJson, hiddenGrade, holm, improvementVerdict, inMemoryReviewStore, inferDomainKeywords, interRaterReliability, interpretCliffs, iqr, isBinaryOutcomeVector, isJudgeSpan, isLlmSpan, isModelPriced, isRunRecord, isToolSpan, isTransientLlmError, jsonShape, jsonlReviewStore, jsonlRunRecordBackend, judgeAgreementView, judgeFamily, judgeSpans, knowledgeReadinessTracePayload, leaderboard, llmJudge, loadScorecard, localCommandRunner, makeFinding, makeProposalFinding, mannWhitneyU, maximumChargeForLlmRequest, mcnemar, mcnemarPower, mcnemarRequiredN, minimumPairsForPairedDeltaTest, mintRolloutRows, modelHasSnapshot, modelPriceKey, mulberry32, notBlocked, objectiveEval, observeAll, otlpTextToTraceAnalysisStore, pairArms, pairRunRecords, pairedBinaryScale, pairedBootstrap, pairedCohensDz, pairedDeltaTest, pairedDeltaTieFraction, pairedEvalueSequence, pairedMde, pairedRiskDifference, pairedRiskDifferenceExact, pairedRiskDifferenceScore, pairedSignTest, pairedTTest, paretoChart, paretoFrontier, parseReflectionResponse, parseRunRecordSafe, parseRuntimeTrajectoryHookEvent, partialCredit, partitionHeldOut, passAtK, pearsonR, preflightModels, probeLlm, productBenchmarkRepoIdentity, profile_exports as profile, projectRuntimeTrajectoryEvidence, proposeSynthesisTargets, ranks, readProductBenchmarkManifest, recordRuns, recordRunsToScorecard, redTeamDataset, redTeamReport, redactString, regexMatches, renderPreferenceMemoryMarkdown, repeatedActionDetector, requiredPairedSampleSize, requiredSampleSize, resolveModelPricing, resolveSeat, roundTripRunRecord, routeFields, runAgentControlLoop, runCampaign, runCanaries, runCounterfactual, runEquivalenceCheck, runEvalCampaign, runIntentMatchJudge, runKeywordCoverageJudge, runKeywordCoverageJudgeUrl, runProposeReview, runProposeReviewAsControlLoop, runSemanticConceptJudge, runTaskScore, runsForScenario, scoreKnowledgeReadiness, scoreRedTeamOutput, scoreTraceInsightReadiness, seatPresets, selfImprove, spearmanR, stripFencedJson, subjectiveEval, summarizeBackendIntegrity, summarizePreferenceMemory, summaryTable, textInSnapshot, toAgentProfileJson, tokenizeDomainWords, toolSpansToTraceAnalysisStore, toolWasteView, traceContract, urlContains, userQuestionsForKnowledgeGaps, validateRunRecord, verbosityBias, verifyAgentProfileCell, verifyCompletion, viteDeployRunner, weightedComposite, weightedMean, wilcoxonSignedRank, wilson, withAssignedFeedbackSplit, withHeldoutBlend, withJudgeRetry, wranglerDeployRunner };
|
|
7406
7406
|
|
|
7407
7407
|
//# sourceMappingURL=index.js.map
|
|
@@ -1,7 +1,7 @@
|
|
|
1
1
|
import { o as FailureClass } from "./schema-BtVldJ3T.js";
|
|
2
|
-
import { l as RunTerminalOutcome } from "./run-record-
|
|
2
|
+
import { l as RunTerminalOutcome } from "./run-record-DVV82Gwh.js";
|
|
3
3
|
import { r as ContinuousAgreement } from "./judge-calibration-C5CbMYce.js";
|
|
4
|
-
import { i as ParetoFigureSpec, t as GainDistributionBin } from "./summary-report-
|
|
4
|
+
import { i as ParetoFigureSpec, t as GainDistributionBin } from "./summary-report-B__Y5ub3.js";
|
|
5
5
|
//#region src/contract/insight-report.d.ts
|
|
6
6
|
interface InsightReport {
|
|
7
7
|
/** Number of runs analyzed. */
|
|
@@ -403,4 +403,4 @@ interface Recommendation {
|
|
|
403
403
|
}
|
|
404
404
|
//#endregion
|
|
405
405
|
export { FailureClusterInsight as a, JudgeInsight as c, Recommendation as d, ReleaseSummary as f, FailureClassTally as i, LiftInsight as l, TokenUsageInsight as m, ExecutionErrorOutcomeCell as n, InsightReport as o, ScalarDistribution as p, ExecutionInsight as r, InterRaterInsight as s, CostProvenanceSummary as t, OutcomeCorrelationInsight as u };
|
|
406
|
-
//# sourceMappingURL=insight-report-
|
|
406
|
+
//# sourceMappingURL=insight-report-BeT8KCgI.d.ts.map
|
|
@@ -1 +1 @@
|
|
|
1
|
-
{"version":3,"file":"insight-report-
|
|
1
|
+
{"version":3,"file":"insight-report-BeT8KCgI.d.ts","names":[],"sources":["../src/contract/insight-report.ts"],"mappings":";;;;;UAqCiB;;EAEf;;;;EAKA,WAAW;;EAGX,WAAW;;;EAIX,cAAc,eAAe;;EAG7B;IACE,MAAM;IACN,QAAQ;;;;IAIR,aAAa;;;;;IAKb;MAAa;MAAe;;;;;;EAM9B,QAAQ,eAAe;;;;EAKvB,aAAa;;;;EAKb,OAAO;;;EAIP,kBAAkB;;;EAIlB,gBAAgB;;;;;EAMhB,qBAAqB;;;;;EAMrB,SAAS;;;;;;EAOT,wBAAwB;;;;EAKxB,iBAAiB;;;EAIjB,iBAAiB;;UAGF;EACf;IAAY;IAAW;;EACvB;IAAa;IAAW;;EACxB;IAAc;;EACd;;UAGe;;EAEf,YAAY;;EAEZ,SAAS;;;EAGT,YAAY;;;EAGZ;IACE;IACA,YAAY;IACZ,SAAS;IACT;;;EAGF,QAAQ;IAAQ;IAAe;;;;;EAI/B;IACE;IACA;IACA;;;;;EAKF;IACE;;;IAGA;;IAEA;;IAEA;;IAEA;;IAEA;;;;;IAKA,mBAAmB,OAAO,oBAAoB;;;;EAIhD;IACE;IACA;IACA;IACA;IACA;;;UAIa;;EAEf;;EAEA;;EAEA;;UAGe;EACf,OAAO;EACP,QAAQ;EACR,WAAW;EACX,QAAQ;EACR,YAAY;EACZ;IACE;IACA;IACA;IACA;IACA;;;;UAOa;;EAEf;;EAEA;;EAEA;;EAEA;;EAEA;;EAEA;;EAEA;;EAEA,WAAW;;;;;EAKX,WAAW;IAAQ;IAAe;;;UAGnB;;EAEf;;EAEA;;;EAGA,cAAc;;;EAGd;;;EAGA;;;EAGA;;UAGe;;EAEf;;EAEA;;EAEA;;EAEA;;EAEA;;EAEA;;EAEA,SAAS;;EAET,mBAAmB;IACjB;IACA,SAAS;MAAQ;MAAe;;IAChC;;;UAIa;EACf;EACA;;EAEA;;EAEA;;;EAGA;;EAEA;;EAEA;;;;;;;;;;;EAWA;;EAEA;;EAEA;;EAEA;;EAEA;;EAEA;;UAGe;;EAEf,UAAU;IACR;IACA;;IAEA;;IAEA;;IAEA;;EAEF;;;;;UAMe;;EAEf,cAAc;;EAEd;;EAEA;;UAGe;;EAEf;;;EAGA;EACA,UAAU;IAAQ;IAAe;IAAgB;;;UAGlC;;;EAGf;;EAEA;;EAEA;;EAEA;;EAEA;IACE;IACA;IACA;;;UAIa;;;EAGf;EACA,MAAM;IACJ;IACA;IACA;;;;EAIF;;UAGQ;;EAER;;EAEA;;;;EAIA;;EAEA;EACA;;KAGU,eACP;EACC;;EAEA;;EAEA;;EAEA;;EAEA;MAED;EACC;;EAEA;EACA;EACA;EACA;;UAGW;;EAEf;EACA;;EAEA;;;;EAIA,SAAS,eAAe;;;EAGxB;;EAEA;;EAEA;;UAGe;EACf;EACA;EACA;EACA;;EAEA"}
|
|
@@ -2,7 +2,7 @@ import { i as JudgeError, s as ValidationError, t as AgentEvalError } from "./er
|
|
|
2
2
|
import { o as weightedComposite, t as confidenceInterval } from "./descriptive-B5MwKfbf.js";
|
|
3
3
|
import { r as pairedBootstrap } from "./paired-tests-BHIhYVdu.js";
|
|
4
4
|
import { n as contentHash, t as canonicalJson } from "./verdict-cache-mZf5FEiY.js";
|
|
5
|
-
import { a as campaignCellCostProvenance, o as campaignCellExecutionEvidence, t as detectRewardHacking, u as projectCampaignCellQuality } from "./reward-hacking-
|
|
5
|
+
import { a as campaignCellCostProvenance, o as campaignCellExecutionEvidence, t as detectRewardHacking, u as projectCampaignCellQuality } from "./reward-hacking-DFo2FU5J.js";
|
|
6
6
|
import { i as CostLedger, t as CostAccountingIncompleteError } from "./cost-ledger-BSe92yAV.js";
|
|
7
7
|
import { u as mapConcurrent } from "./ledger-core-BmZt19oQ.js";
|
|
8
8
|
import { b as createRunCostLedger, h as isRecord, m as isExternalTextCandidate, x as fsCampaignStorage } from "./external-optimizer-subprocess-BhKYK0Jv.js";
|
|
@@ -4861,4 +4861,4 @@ function firstString(value) {
|
|
|
4861
4861
|
//#endregion
|
|
4862
4862
|
export { assertCampaignDesign as $, surfaceHash as A, DEFAULT_RED_TEAM_CORPUS as B, optimizationTokenUsageFromSummary as C, componentSurfaceIdentityMaterial as D, codeSurfaceIdentityMaterial as E, parseReflectionResponse as F, runEval as G, redTeamReport as H, recoverTruncatedJson as I, resolveRunDir as J, runCampaign as K, assertGepaCandidatePopulationSummary as L, campaignMeanComposite as M, compareRankKeys as N, renderSurfaceDiff as O, buildReflectionPrompt as P, summarizeBackendIntegrity as Q, readGepaCandidatePopulationArtifact as R, costFromLedgerSummary as S, assertComponentSurface as T, scoreRedTeamOutput as U, redTeamDataset as V, runCanaries as W, BackendIntegrityError as X, tangleTracesRoot as Y, assertRealBackend as Z, paretoFrontier as _, canonicalDigest as a, computeManifestHash as at, combineComparisonCosts as b, loopProvenanceSpans as c, verifyLoopProvenanceRecord as d, assertCampaignSplitIdentity as et, runImprovementLoop as f, dominates as g, labelTrustRank as h, campaignMeasurementDigest as i, cellCachePath as it, campaignBreakdown as j, surfaceContentHash as k, provenanceRecordPath as l, isProposedCandidate as m, JudgeParseError as n, campaignSplitDigest as nt, emitLoopProvenance as o, runOptimization as p, planCampaignRun as q, buildLoopProvenanceRecord as r, campaignSplitDigestFromIdentities as rt, loopProvenanceArgsFromResult as s, llmJudge as t, campaignScenarioIdentity as tt, provenanceSpansPath as u, openAutoPr as v, assertCodeSurfaceIdentity as w, compareOptimizationMethods as x, assertOptimizationResult as y, defaultProductionGate as z };
|
|
4863
4863
|
|
|
4864
|
-
//# sourceMappingURL=llm-judge-
|
|
4864
|
+
//# sourceMappingURL=llm-judge-BO1LGjdn.js.map
|