@tangle-network/agent-eval 0.94.0 → 0.95.1
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +32 -0
- package/README.md +44 -30
- package/dist/adapters/http.d.ts +8 -7
- package/dist/adapters/http.js.map +1 -1
- package/dist/adapters/langchain.d.ts +3 -2
- package/dist/adapters/otel.d.ts +5 -4
- package/dist/analyst/index.d.ts +11 -31
- package/dist/analyst/index.js +5 -65
- package/dist/analyst/index.js.map +1 -1
- package/dist/{analyze-runs-B6Ljo_dI.d.ts → analyze-runs-DtT6F_6T.d.ts} +3 -3
- package/dist/belief-state/index.d.ts +4 -3
- package/dist/benchmarks/index.d.ts +3 -2
- package/dist/campaign/index.d.ts +727 -616
- package/dist/campaign/index.js +1863 -1316
- package/dist/campaign/index.js.map +1 -1
- package/dist/{chunk-2K6UUZ7P.js → chunk-2T4EZACH.js} +1 -1
- package/dist/chunk-2T4EZACH.js.map +1 -0
- package/dist/{chunk-CTBHKLEU.js → chunk-77T4STFI.js} +59 -86
- package/dist/chunk-77T4STFI.js.map +1 -0
- package/dist/{chunk-EGPMSBEZ.js → chunk-7QTQKIDD.js} +178 -177
- package/dist/chunk-7QTQKIDD.js.map +1 -0
- package/dist/{chunk-MIFZUPEK.js → chunk-AQ5WQAIV.js} +21 -6
- package/dist/chunk-AQ5WQAIV.js.map +1 -0
- package/dist/chunk-DJWX3GVS.js +81 -0
- package/dist/chunk-DJWX3GVS.js.map +1 -0
- package/dist/{chunk-S6OZEZQK.js → chunk-HMA63UEO.js} +37 -9
- package/dist/{chunk-S6OZEZQK.js.map → chunk-HMA63UEO.js.map} +1 -1
- package/dist/{chunk-TBDR6PAI.js → chunk-IZCEK2HR.js} +2 -2
- package/dist/{chunk-SD2YFWQQ.js → chunk-KKWJD5E6.js} +20 -20
- package/dist/chunk-KKWJD5E6.js.map +1 -0
- package/dist/{chunk-KWRRMR3J.js → chunk-LO6IOIJ2.js} +10 -10
- package/dist/chunk-LO6IOIJ2.js.map +1 -0
- package/dist/{chunk-E4GH6USR.js → chunk-NZEQVRH5.js} +2 -2
- package/dist/chunk-NZEQVRH5.js.map +1 -0
- package/dist/{chunk-MPQWFX6Y.js → chunk-PSWWQXHF.js} +13 -88
- package/dist/chunk-PSWWQXHF.js.map +1 -0
- package/dist/{chunk-Q5LIB7BC.js → chunk-S4SYLDFX.js} +2 -2
- package/dist/chunk-S4SYLDFX.js.map +1 -0
- package/dist/{chunk-KW53MSA5.js → chunk-X74V6ESX.js} +2 -2
- package/dist/{chunk-QMUEXQJS.js → chunk-YBIGNSCZ.js} +81 -4
- package/dist/chunk-YBIGNSCZ.js.map +1 -0
- package/dist/{chunk-2KNZHH3P.js → chunk-Z6L6YSU6.js} +2 -2
- package/dist/{code-agent-session-BO8nCnv3.d.ts → code-agent-session-CPHRCb4-.d.ts} +1 -1
- package/dist/contract/index.d.ts +91 -43
- package/dist/contract/index.js +127 -17
- package/dist/contract/index.js.map +1 -1
- package/dist/{control-D6qwHXIR.d.ts → control-Doncu-B_.d.ts} +2 -2
- package/dist/control.d.ts +3 -2
- package/dist/control.js +2 -2
- package/dist/{corpus-B8A4BDR3.d.ts → corpus-D4YW9UoJ.d.ts} +1 -1
- package/dist/{default-registry-6dhErQbs.d.ts → default-registry-GyE8X5SP.d.ts} +3 -3
- package/dist/diagnose.d.ts +4 -3
- package/dist/diagnose.js +1 -1
- package/dist/{run-improvement-loop-DBahB8Ax.d.ts → gepa-C1NCIZ9o.d.ts} +117 -130
- package/dist/hosted/index.d.ts +5 -4
- package/dist/{index-Bx3gZ8xl.d.ts → index-_Y4oNOOb.d.ts} +1 -1
- package/dist/index.d.ts +76 -81
- package/dist/index.js +66 -31
- package/dist/index.js.map +1 -1
- package/dist/{insight-report-DWl3z9tl.d.ts → insight-report-BnRjTibG.d.ts} +1 -1
- package/dist/{kind-factory-0BhLSI27.d.ts → kind-factory-X3eDYbKn.d.ts} +2 -3
- package/dist/matrix/index.d.ts +1 -1
- package/dist/meta-eval/index.d.ts +3 -2
- package/dist/multishot/index.d.ts +4 -4
- package/dist/multishot/index.js.map +1 -1
- package/dist/openapi.json +1 -1
- package/dist/{pre-registration-mAnCugl9.d.ts → pre-registration-nfUdc9EQ.d.ts} +2 -42
- package/dist/{provenance-P-bCL2Fo.d.ts → provenance-CncDq9qE.d.ts} +26 -41
- package/dist/{release-report-BEbWmVYj.d.ts → release-report-pidWUMZ2.d.ts} +2 -2
- package/dist/reporting.d.ts +5 -4
- package/dist/{researcher-B0C2_fVO.d.ts → researcher-Jr8ME1dZ.d.ts} +2 -2
- package/dist/rl.d.ts +516 -515
- package/dist/rl.js +612 -612
- package/dist/rl.js.map +1 -1
- package/dist/{rubric-predictive-validity-Cy_W-hWZ.d.ts → rubric-predictive-validity-C2hDKM8Z.d.ts} +1 -1
- package/dist/{run-campaign-7WNXMDSN.js → run-campaign-WXY7KI67.js} +2 -2
- package/dist/{run-record-e7vj1uZQ.d.ts → run-record-CP2ObebC.d.ts} +14 -18
- package/dist/{runtime-trajectory-BDgfGZSr.d.ts → runtime-trajectory-BOUUjI0y.d.ts} +1 -1
- package/dist/{semantic-concept-judge-B9MgmBnM.d.ts → semantic-concept-judge-DSBB2Cfp.d.ts} +2 -2
- package/dist/{summary-report-BDOFevaT.d.ts → summary-report-CInXwsza.d.ts} +1 -1
- package/dist/testing-C21CHsq2.d.ts +20 -0
- package/dist/testing.d.ts +1 -0
- package/dist/testing.js +8 -0
- package/dist/testing.js.map +1 -0
- package/dist/traces.d.ts +26 -10
- package/dist/traces.js +41 -11
- package/dist/{types-Ce17tDlG.d.ts → types-B5x54y6n.d.ts} +1 -1
- package/dist/{types-mn5Aqk7x.d.ts → types-BUxNaJ8c.d.ts} +2 -4
- package/dist/{types-BU-7W85F.d.ts → types-DQRY8ZT-.d.ts} +60 -58
- package/dist/workflow/index.d.ts +5 -4
- package/dist/workflow/index.js +1 -1
- package/docs/campaign-proposers.md +170 -0
- package/docs/concepts.md +8 -4
- package/docs/customer-journeys.md +15 -13
- package/docs/design/loop-taxonomy.md +34 -66
- package/docs/distributed-driver.md +14 -14
- package/docs/feature-guide.md +1 -1
- package/docs/hosted-ingest-spec.md +2 -3
- package/docs/multi-shot-optimization.md +8 -8
- package/docs/product-eval-adoption.md +1 -1
- package/docs/self-improvement-map.md +33 -29
- package/package.json +8 -14
- package/dist/chunk-2K6UUZ7P.js.map +0 -1
- package/dist/chunk-CTBHKLEU.js.map +0 -1
- package/dist/chunk-E4GH6USR.js.map +0 -1
- package/dist/chunk-EGPMSBEZ.js.map +0 -1
- package/dist/chunk-KWRRMR3J.js.map +0 -1
- package/dist/chunk-MIFZUPEK.js.map +0 -1
- package/dist/chunk-MPQWFX6Y.js.map +0 -1
- package/dist/chunk-Q5LIB7BC.js.map +0 -1
- package/dist/chunk-QMUEXQJS.js.map +0 -1
- package/dist/chunk-SD2YFWQQ.js.map +0 -1
- package/docs/design/external-agent-wedge.md +0 -89
- package/docs/design/phase-d-rfc.md +0 -125
- package/docs/design/phase4-consumer-migration.md +0 -70
- package/docs/design/primitives-integration-spec.md +0 -393
- package/docs/design/product-self-improvement-loop.md +0 -146
- package/docs/design/self-improvement-engine.md +0 -140
- package/docs/design/self-improvement-protocol.md +0 -223
- package/docs/design/self-improvement-roadmap.md +0 -106
- package/docs/design/substrate-gaps.md +0 -118
- package/docs/phase-b-pairing-kit.md +0 -188
- package/docs/phase-b-runbook.md +0 -176
- package/docs/pilot/README.md +0 -62
- package/docs/pilot/customer-checklist.md +0 -90
- package/docs/pilot/integration-foreign-stack.md +0 -296
- package/docs/pilot/integration-tangle-stack.md +0 -248
- package/docs/pilot/one-pager.md +0 -161
- package/docs/pilot/sample-insight-report.json +0 -172
- package/docs/quickstart-external.md +0 -229
- package/docs/research/belief-state-agent-eval-roadmap.md +0 -593
- package/docs/research/research-roadmap.md +0 -205
- package/docs/specs/driver-honest-spec.md +0 -251
- package/docs/specs/hermes-self-improvement-audit.md +0 -93
- package/docs/specs/profile-versioning.md +0 -291
- package/docs/three-package-architecture.md +0 -168
- /package/dist/{chunk-TBDR6PAI.js.map → chunk-IZCEK2HR.js.map} +0 -0
- /package/dist/{chunk-KW53MSA5.js.map → chunk-X74V6ESX.js.map} +0 -0
- /package/dist/{chunk-2KNZHH3P.js.map → chunk-Z6L6YSU6.js.map} +0 -0
- /package/dist/{run-campaign-7WNXMDSN.js.map → run-campaign-WXY7KI67.js.map} +0 -0
package/dist/index.d.ts
CHANGED
|
@@ -1,11 +1,11 @@
|
|
|
1
|
-
export { A as ActionExecutionPolicy, a as ActionPolicyDecision, C as ControlRunToRunRecordOptions, L as LlmJsonCall, b as LlmReviewerConfig, P as ProposeFn, c as ProposeInput, d as ProposeOutput, e as ProposeReviewConfig, f as ProposeReviewControlAction, g as ProposeReviewControlConfig, h as ProposeReviewControlResult, i as ProposeReviewControlState, j as ProposeReviewReport, k as ProposeReviewShot, R as Review, l as ReviewFn, m as ReviewInput, n as ReviewMemoryEntry, o as ReviewMemoryStore, p as RunEvidenceMetadata, V as Verification, q as VerifyFn, r as controlFailureClassFromVerification, s as controlRunToRunRecord, t as createLlmReviewer, u as evaluateActionPolicy, v as inMemoryReviewStore, w as jsonlReviewStore, x as runProposeReview, y as runProposeReviewAsControlLoop, z as scoreFromEvals } from './control-
|
|
2
|
-
import { R as RunRecord, a as RunSplitTag } from './run-record-
|
|
3
|
-
export { e as AGENT_PROFILE_KINDS, A as AgentProfileCell, d as AgentProfileCellInput,
|
|
4
|
-
export { B as BehavioralMetrics, x as ConceptComplexity, y as ConceptFinding, z as ConceptSpec, A as ConceptWeightStrategy, C as CreateAnalystAiConfig, E as DEFAULT_COMPLEXITY_WEIGHTS, D as DEFAULT_TRACE_ANALYST_KINDS, b as DiffPolicy, F as FAILURE_MODE_KIND_SPEC, e as FindingSubject, f as FindingSubjectKind, h as FindingsDiff, i as FindingsStore, I as IMPROVEMENT_KIND_SPEC, j as KNOWLEDGE_GAP_KIND_SPEC, k as KNOWLEDGE_POISONING_KIND_SPEC, P as PersistedFinding, G as SEMANTIC_CONCEPT_JUDGE_VERSION, l as SKILL_USAGE_ANALYST, a as SemanticConceptJudgeInput, S as SemanticConceptJudgeOptions, H as SemanticConceptJudgeResult, m as SkillUsageAnalyst, J as SuboptimalCode, L as SuboptimalSignal, M as computeTraceMetrics, r as createAnalystAi, N as createSemanticConceptJudge, s as defaultIsMaterial, t as diffFindings, O as runSemanticConceptJudge } from './semantic-concept-judge-
|
|
5
|
-
import { l as ChatRequest, p as CreateChatClientOpts } from './types-
|
|
6
|
-
export { A as Analyst, a as AnalystContext, g as AnalystCost, c as AnalystFinding, i as AnalystInputKind, j as AnalystRequirements, f as AnalystRunEvent, e as AnalystRunInputs, d as AnalystRunResult, b as AnalystRunSummary, h as AnalystSeverity, k as ChatCallOpts, C as ChatClient, m as ChatResponse, n as ChatTransport, o as CliBridgeTransportOpts, D as DirectProviderTransportOpts, E as EvidenceRef, M as MockTransportOpts, R as RouterTransportOpts, S as SandboxSdkTransportOpts, q as computeFindingId, r as createChatClient, s as makeFinding } from './types-
|
|
7
|
-
export { a as AnalystHooks, A as AnalystRegistry,
|
|
8
|
-
export { C as CreateTraceAnalystKindOpts, a as RawAnalystFinding, c as TraceAnalystGolden, T as TraceAnalystKindSpec, d as createTraceAnalystKind, r as renderPriorFindings } from './kind-factory-
|
|
1
|
+
export { A as ActionExecutionPolicy, a as ActionPolicyDecision, C as ControlRunToRunRecordOptions, L as LlmJsonCall, b as LlmReviewerConfig, P as ProposeFn, c as ProposeInput, d as ProposeOutput, e as ProposeReviewConfig, f as ProposeReviewControlAction, g as ProposeReviewControlConfig, h as ProposeReviewControlResult, i as ProposeReviewControlState, j as ProposeReviewReport, k as ProposeReviewShot, R as Review, l as ReviewFn, m as ReviewInput, n as ReviewMemoryEntry, o as ReviewMemoryStore, p as RunEvidenceMetadata, V as Verification, q as VerifyFn, r as controlFailureClassFromVerification, s as controlRunToRunRecord, t as createLlmReviewer, u as evaluateActionPolicy, v as inMemoryReviewStore, w as jsonlReviewStore, x as runProposeReview, y as runProposeReviewAsControlLoop, z as scoreFromEvals } from './control-Doncu-B_.js';
|
|
2
|
+
import { R as RunRecord, a as RunSplitTag } from './run-record-CP2ObebC.js';
|
|
3
|
+
export { e as AGENT_PROFILE_KINDS, f as AgentInterfaceProfileLike, A as AgentProfileCell, d as AgentProfileCellInput, g as AgentProfileCellSchemaVersion, h as AgentProfileCellValidationError, i as AgentProfileDimensionValue, j as AgentProfileHarness, k as AgentProfileJson, l as AgentProfileKind, m as AgentProfileSource, n as AgentProfileSourceInput, J as JudgeScoresRecord, c as RunJudgeMetadata, o as RunOutcome, p as RunRecordValidationError, b as RunTokenUsage, q as agentProfileCellHashMaterial, r as agentProfileCellKey, s as assertRunAgentProfileCell, t as buildAgentInterfaceProfileCell, u as buildAgentProfileCell, v as groupRunsByAgentProfileCell, w as isRunRecord, x as parseRunRecordSafe, y as requireAgentProfileCell, z as roundTripRunRecord, B as toAgentProfileJson, C as validateAgentProfileCell, D as validateRunRecord, E as verifyAgentProfileCell } from './run-record-CP2ObebC.js';
|
|
4
|
+
export { B as BehavioralMetrics, x as ConceptComplexity, y as ConceptFinding, z as ConceptSpec, A as ConceptWeightStrategy, C as CreateAnalystAiConfig, E as DEFAULT_COMPLEXITY_WEIGHTS, D as DEFAULT_TRACE_ANALYST_KINDS, b as DiffPolicy, F as FAILURE_MODE_KIND_SPEC, e as FindingSubject, f as FindingSubjectKind, h as FindingsDiff, i as FindingsStore, I as IMPROVEMENT_KIND_SPEC, j as KNOWLEDGE_GAP_KIND_SPEC, k as KNOWLEDGE_POISONING_KIND_SPEC, P as PersistedFinding, G as SEMANTIC_CONCEPT_JUDGE_VERSION, l as SKILL_USAGE_ANALYST, a as SemanticConceptJudgeInput, S as SemanticConceptJudgeOptions, H as SemanticConceptJudgeResult, m as SkillUsageAnalyst, J as SuboptimalCode, L as SuboptimalSignal, M as computeTraceMetrics, r as createAnalystAi, N as createSemanticConceptJudge, s as defaultIsMaterial, t as diffFindings, O as runSemanticConceptJudge } from './semantic-concept-judge-DSBB2Cfp.js';
|
|
5
|
+
import { l as ChatRequest, p as CreateChatClientOpts } from './types-B5x54y6n.js';
|
|
6
|
+
export { A as Analyst, a as AnalystContext, g as AnalystCost, c as AnalystFinding, i as AnalystInputKind, j as AnalystRequirements, f as AnalystRunEvent, e as AnalystRunInputs, d as AnalystRunResult, b as AnalystRunSummary, h as AnalystSeverity, k as ChatCallOpts, C as ChatClient, m as ChatResponse, n as ChatTransport, o as CliBridgeTransportOpts, D as DirectProviderTransportOpts, E as EvidenceRef, M as MockTransportOpts, R as RouterTransportOpts, S as SandboxSdkTransportOpts, q as computeFindingId, r as createChatClient, s as makeFinding } from './types-B5x54y6n.js';
|
|
7
|
+
export { a as AnalystHooks, A as AnalystRegistry, b as AnalystRegistryOptions, B as BudgetPolicy, D as DefaultAnalystRegistryOptions, R as RegistryRunOpts, c as buildDefaultAnalystRegistry } from './default-registry-GyE8X5SP.js';
|
|
8
|
+
export { C as CreateTraceAnalystKindOpts, a as RawAnalystFinding, c as TraceAnalystGolden, T as TraceAnalystKindSpec, d as createTraceAnalystKind, r as renderPriorFindings } from './kind-factory-X3eDYbKn.js';
|
|
9
9
|
import { TCloud } from '@tangle-network/tcloud';
|
|
10
10
|
import { B as BenchmarkRunnerConfig, S as Scenario, c as BenchmarkReport, P as ProductClientConfig, C as CheckResult, T as TestResult, d as PersonaConfig, D as DriverResult, e as DriverState, b as JudgeFn, f as CollectedArtifacts, g as ScenarioResult, h as TurnMetrics, i as ScenarioFile, j as CompletionCriterion } from './types-C7DGg5ex.js';
|
|
11
11
|
export { A as ArtifactCheck, k as ArtifactResult, E as EvalResult, F as FeedbackPattern, l as JudgeConfig, a as JudgeInput, m as JudgeRubric, J as JudgeScore, n as PersonaRigor, R as RouteMap, o as RubricDimension, p as Turn, q as TurnResult } from './types-C7DGg5ex.js';
|
|
@@ -16,19 +16,19 @@ import { A as AgentEvalError, J as JudgeError, a as ConfigError } from './errors
|
|
|
16
16
|
export { b as AgentEvalErrorCode, C as CaptureIntegrityError, N as NotFoundError, R as ReplayError, V as ValidationError, c as VerificationError } from './errors-CzMUYo7b.js';
|
|
17
17
|
import { b as FeedbackLabel, F as FeedbackTrajectoryStore, a as FeedbackTrajectory } from './feedback-trajectory-BxY0cKfs.js';
|
|
18
18
|
export { c as FeedbackArtifactType, d as FeedbackAttempt, e as FeedbackLabelKind, f as FeedbackLabelSource, g as FeedbackOptimizerRow, h as FeedbackOutcome, i as FeedbackReplayAdapter, j as FeedbackReplayResult, k as FeedbackSeverity, l as FeedbackSplitPolicy, m as FeedbackTask, n as FeedbackTrajectoryFilter, o as FileSystemFeedbackTrajectoryStore, I as InMemoryFeedbackTrajectoryStore, P as PreferenceMemoryEntry, p as ProposedSideEffect, q as assignFeedbackSplit, r as controlRunToFeedbackTrajectory, s as createFeedbackTrajectory, t as feedbackTrajectoriesToDatasetScenarios, u as feedbackTrajectoriesToOptimizerRows, v as feedbackTrajectoryToDatasetScenario, w as feedbackTrajectoryToOptimizerRow, x as parseFeedbackTrajectoriesJsonl, y as renderPreferenceMemoryMarkdown, z as replayFeedbackTrajectories, A as replayFeedbackTrajectory, B as serializeFeedbackTrajectoriesJsonl, C as summarizePreferenceMemory, D as withAssignedFeedbackSplit } from './feedback-trajectory-BxY0cKfs.js';
|
|
19
|
-
import { b as CorrectnessChecker
|
|
20
|
-
export {
|
|
19
|
+
import { b as CorrectnessChecker } from './pre-registration-nfUdc9EQ.js';
|
|
20
|
+
export { A as ArtifactCheckArtifact, c as ArtifactEventLike, d as ArtifactValidator, e as BackendIntegrityError, B as BackendIntegrityReport, C as CompletionRequirement, a as CompletionVerdict, H as HypothesisManifest, f as HypothesisResult, L as LlmCorrectnessCheckerOpts, g as ProducedProposal, P as ProducedState, h as ProposalEventLike, i as RequirementCheck, R as RuntimeEventLike, j as SatisfiedBy, S as SignedManifest, k as SignedManifestAlgo, T as TaskGold, l as ToolCallEventLike, V as ValidationContext, m as ValidationIssue, n as ValidationResult, o as assertRealBackend, p as byteLengthRange, q as canonicalize, r as completionVerdict, s as composeValidators, t as containsAll, u as createLlmCorrectnessChecker, v as createTokenRecallChecker, w as evaluateHypothesis, x as extractProducedState, y as hashJson, z as jsonHasKeys, D as parseCorrectnessResponse, E as regexMatch, F as signManifest, G as summarizeBackendIntegrity, I as verifyCompletion, J as verifyManifest } from './pre-registration-nfUdc9EQ.js';
|
|
21
21
|
export { DataAcquisitionPlan, KnowledgeAcquisitionMode, KnowledgeBundle, KnowledgeFallbackPolicy, KnowledgeFreshness, KnowledgeImportance, KnowledgeReadinessReport, KnowledgeRecommendedAction, KnowledgeRequirement, KnowledgeRequirementCategory, KnowledgeResponsibleSurface, KnowledgeSensitivity, ScoreKnowledgeReadinessOptions, UserQuestion, acquisitionPlansForKnowledgeGaps, blockingKnowledgeEval, knowledgeReadinessTracePayload, scoreKnowledgeReadiness, userQuestionsForKnowledgeGaps } from './knowledge/index.js';
|
|
22
|
-
import { h as ReleaseConfidenceThresholds, f as ReleaseConfidenceScorecard } from './release-report-
|
|
23
|
-
export { A as ActionableSideInfo, o as AsiSeverity, B as BootstrapOptions, a as BootstrapResult, J as JudgeReplayGateArgs, R as ReleaseConfidenceAxis, b as ReleaseConfidenceAxisName, c as ReleaseConfidenceInput, d as ReleaseConfidenceIssue, e as ReleaseConfidenceMetrics, g as ReleaseConfidenceStatus, i as ReleaseTraceEvidence, j as RenderReleaseReportOptions, V as Verdict, k as assertReleaseConfidence, l as bootstrapCi, m as evaluateReleaseConfidence, n as judgeReplayGate, r as renderReleaseReport } from './release-report-
|
|
22
|
+
import { h as ReleaseConfidenceThresholds, f as ReleaseConfidenceScorecard } from './release-report-pidWUMZ2.js';
|
|
23
|
+
export { A as ActionableSideInfo, o as AsiSeverity, B as BootstrapOptions, a as BootstrapResult, J as JudgeReplayGateArgs, R as ReleaseConfidenceAxis, b as ReleaseConfidenceAxisName, c as ReleaseConfidenceInput, d as ReleaseConfidenceIssue, e as ReleaseConfidenceMetrics, g as ReleaseConfidenceStatus, i as ReleaseTraceEvidence, j as RenderReleaseReportOptions, V as Verdict, k as assertReleaseConfidence, l as bootstrapCi, m as evaluateReleaseConfidence, n as judgeReplayGate, r as renderReleaseReport } from './release-report-pidWUMZ2.js';
|
|
24
24
|
export { c as CliffsMagnitude, d as CorpusAgreementOptions, e as CorpusAgreementPerDimension, C as CorpusAgreementReport, f as CorpusScoreRecord, g as EProcess, h as EProcessOptions, E as EProcessState, i as EProcessStep, M as McNemarResult, P as PairedBootstrapOptions, a as PairedBootstrapResult, j as ProportionInterval, R as RiskDifferenceResult, W as WeightedCompositeInput, k as WeightedCompositeResult, b as benjaminiHochberg, l as bonferroni, m as cliffsDelta, n as cohensD, o as confidenceInterval, q as corpusInterRaterAgreement, r as corpusInterRaterAgreementFromJudgeScores, s as eProcess, t as interRaterReliability, u as interpretCliffs, v as mannWhitneyU, x as mcnemar, y as mcnemarPower, z as mcnemarRequiredN, A as mulberry32, B as normalizeScores, p as pairedBootstrap, D as pairedMde, F as pairedRiskDifference, G as pairedTTest, H as partialCredit, I as passAtK, J as requiredSampleSize, K as weightedComposite, L as weightedMean, w as wilcoxonSignedRank, N as wilson } from './statistics-CCJpTGOS.js';
|
|
25
|
+
import { OtelExporter, OtelExportConfig } from './traces.js';
|
|
26
|
+
export { CaptureFetchContext, CaptureFetchOptions, ExportableSpan, ExtractedUsage, FlattenOtlpOptions, LLM_CACHED_TOKENS, LLM_CACHED_TOKEN_ATTR_KEYS, LLM_COST_ATTR_KEYS, LLM_COST_USD, LLM_INPUT_TOKENS, LLM_INPUT_TOKEN_ATTR_KEYS, LLM_MODEL_ATTR_KEYS, LLM_MODEL_NAME, LLM_OUTPUT_TOKENS, LLM_OUTPUT_TOKEN_ATTR_KEYS, OPENINFERENCE_SPAN_KIND, OTEL_AGENT_EVAL_SCOPE, OtlpExport, OtlpFileTraceStore, OtlpFileTraceStoreOptions, OtlpFlatLine, OtlpResourceSpans, OtlpSpan, OtlpToRunRecordsOptions, OtlpTraceRunRecord, ProjectedOtlpSpan, ReplayCache, ReplayCacheEntry, ReplayCacheMissError, ReplayCacheStats, ReplayFetchOptions, SPAN_KIND_ATTR_KEYS, SpanNotFoundError, TOOL_NAME, TOOL_NAME_ATTR_KEYS, TRACE_ANALYST_ACTOR_DESCRIPTION, TRACE_ANALYST_ACTOR_DESCRIPTION_VERSION, TRACE_ANALYST_SUBAGENT_DESCRIPTION, TraceAggregate, TraceAnalystHookOptions, TraceFileMissingError, TraceInsightContext, TraceInsightFinding, TraceInsightPanelRole, TraceInsightPromptInput, TraceInsightQualityGate, TraceInsightQuestion, TraceInsightReadiness, TraceInsightSuite, TraceInsightTask, TraceNotFoundError, TraceStoreSource, TraceStoreToOtlpOptions, TracesToOtlpResult, asNumber, asString, buildTraceAnalystTools, buildTraceInsightContext, buildTraceInsightPrompt, captureFetchToRawSink, convertTraceStoresToOtlp, createOtelExporter, createOtelTracingStore, createReplayFetch, defaultTraceInsightPanel, describeTraceInsightScope, domainEvidencePattern, exportRunAsOtlp, extractOtlpAttributes, extractUsage, extractUsageFromResponse, extractUsageFromSse, firstNumberAttr, firstStringAttr, flattenOtlpExportToNdjson, inferDomainKeywords, inferOtlpKind, iterateRawCalls, otelRunCompleteHook, otlpToRunRecords, otlpToTraceRunRecords, planTraceInsightQuestions, projectOtlpFlatLine, readOtlpStatus, scoreTraceInsightReadiness, stringField, tokenizeDomainWords, traceAnalystFunctionGroup, traceAnalystOnRunComplete, traceSpanKindToOpenInferenceKind } from './traces.js';
|
|
25
27
|
import { a as AnalyzeTracesInput, A as AnalyzeTracesOptions, b as AnalyzeTracesResult } from './analyst-C8HHvfJp.js';
|
|
26
28
|
export { c as AnalyzeTracesTurnSnapshot, d as analyzeTraces } from './analyst-C8HHvfJp.js';
|
|
27
|
-
import { OtelExporter, OtelExportConfig } from './traces.js';
|
|
28
|
-
export { CaptureFetchContext, CaptureFetchOptions, ExportableSpan, ExtractedUsage, FlattenOtlpOptions, OTEL_AGENT_EVAL_SCOPE, OtlpExport, OtlpFileTraceStore, OtlpFileTraceStoreOptions, OtlpFlatLine, OtlpResourceSpans, OtlpSpan, OtlpToRunRecordsOptions, OtlpTraceRunRecord, ProjectedOtlpSpan, ReplayCache, ReplayCacheEntry, ReplayCacheMissError, ReplayCacheStats, ReplayFetchOptions, SpanNotFoundError, TRACE_ANALYST_ACTOR_DESCRIPTION, TRACE_ANALYST_ACTOR_DESCRIPTION_VERSION, TRACE_ANALYST_SUBAGENT_DESCRIPTION, TraceAggregate, TraceAnalystHookOptions, TraceFileMissingError, TraceInsightContext, TraceInsightFinding, TraceInsightPanelRole, TraceInsightPromptInput, TraceInsightQualityGate, TraceInsightQuestion, TraceInsightReadiness, TraceInsightSuite, TraceInsightTask, TraceNotFoundError, TraceStoreSource, TraceStoreToOtlpOptions, TracesToOtlpResult, asNumber, asString, buildTraceAnalystTools, buildTraceInsightContext, buildTraceInsightPrompt, captureFetchToRawSink, convertTraceStoresToOtlp, createOtelExporter, createOtelTracingStore, createReplayFetch, defaultTraceInsightPanel, describeTraceInsightScope, domainEvidencePattern, exportRunAsOtlp, extractOtlpAttributes, extractUsage, extractUsageFromResponse, extractUsageFromSse, firstNumberAttr, firstStringAttr, flattenOtlpExportToNdjson, inferDomainKeywords, inferOtlpKind, iterateRawCalls, otelRunCompleteHook, otlpToRunRecords, otlpToTraceRunRecords, planTraceInsightQuestions, projectOtlpFlatLine, readOtlpStatus, scoreTraceInsightReadiness, stringField, tokenizeDomainWords, traceAnalystFunctionGroup, traceAnalystOnRunComplete } from './traces.js';
|
|
29
29
|
export { D as DEFAULT_TRACE_ANALYST_BUDGETS, b as DatasetOverview, E as ErrorCluster, Q as QueryTracesPage, S as SearchSpanResult, c as SearchTraceResult, d as SpanMatchRecord, e as TRACE_ANALYST_TRUNCATION_MARKER_PREFIX, T as TraceAnalysisStore, f as TraceAnalystByteBudgets, g as TraceAnalystFilters, a as TraceAnalystSpan, h as TraceAnalystSpanKind, i as TraceAnalystSpanStatus, j as TraceAnalystTraceSummary, V as ViewSpansResult, k as ViewTraceOversized, l as ViewTraceResult } from './store-C1YxJDEK.js';
|
|
30
|
-
import { a as JudgeConfig, S as Scenario$1, G as Gate, J as JudgeScore } from './types-
|
|
31
|
-
import { A as AnalyzeRunsOptions } from './analyze-runs-
|
|
30
|
+
import { a as JudgeConfig, S as Scenario$1, G as Gate, J as JudgeScore } from './types-DQRY8ZT-.js';
|
|
31
|
+
import { A as AnalyzeRunsOptions } from './analyze-runs-DtT6F_6T.js';
|
|
32
32
|
import { S as SteeringBundle } from './harness-optimizer-mOl9XX_O.js';
|
|
33
33
|
export { D as DEFAULT_HARNESS_OBJECTIVES, H as HarnessAdapter, a as HarnessExperimentConfig, b as HarnessExperimentResult, c as HarnessIntervention, d as HarnessRunRequest, e as HarnessRunResult, f as HarnessScenario, g as HarnessSelection, h as HarnessVariant, i as HarnessVariantReport, M as MeasurementPolicy, j as SteeringDelta, k as SteeringRolePrompt, W as WorkflowTopology, m as mergeSteeringBundle, r as renderSteeringText, l as runHarnessExperiment, s as selectHarnessVariant, n as summarizeHarnessResults } from './harness-optimizer-mOl9XX_O.js';
|
|
34
34
|
import { S as SandboxDriver, H as HarnessConfig, a as SandboxHarnessResult } from './test-graded-scenario-DeODGLra.js';
|
|
@@ -44,11 +44,13 @@ export { D as DEFAULT_REDACTION_RULES, b as REDACTION_VERSION, a as RedactionRep
|
|
|
44
44
|
import { T as TraceStore, R as RunFilter } from './store-BcFXE6LG.js';
|
|
45
45
|
export { E as EventFilter, F as FileSystemTraceStore, a as FileSystemTraceStoreOptions, I as InMemoryTraceStore, S as SpanFilter } from './store-BcFXE6LG.js';
|
|
46
46
|
export { D as DEFAULT_FAILURE_RULES, b as FailureClassification, c as FailureContext, d as FailureRule, e as classifyFailure } from './failure-cluster-DH9Flgcf.js';
|
|
47
|
-
export { P as ProjectRuntimeTrajectoryEvidenceOptions, a as RuntimeTrajectoryEvidenceProjection, b as RuntimeTrajectoryEvidenceSummary, c as RuntimeTrajectoryHookEvent, R as RuntimeTrajectoryRecord, d as RuntimeTrajectoryRunRecord, p as parseRuntimeTrajectoryHookEvent, e as projectRuntimeTrajectoryEvidence } from './runtime-trajectory-
|
|
47
|
+
export { P as ProjectRuntimeTrajectoryEvidenceOptions, a as RuntimeTrajectoryEvidenceProjection, b as RuntimeTrajectoryEvidenceSummary, c as RuntimeTrajectoryHookEvent, R as RuntimeTrajectoryRecord, d as RuntimeTrajectoryRunRecord, p as parseRuntimeTrajectoryHookEvent, e as projectRuntimeTrajectoryEvidence } from './runtime-trajectory-BOUUjI0y.js';
|
|
48
48
|
import { a as BaselineReport } from './baseline-Bbid3WoO.js';
|
|
49
49
|
export { B as BaselineOptions, M as MetricSamples, b as MetricVerdict, T as ToolStats, d as ToolUseMetrics, e as ToolUseOptions, f as compareToBaseline, c as computeToolUseMetrics, i as iqr, w as welchsTTest } from './baseline-Bbid3WoO.js';
|
|
50
50
|
import { a as TrajectoryStep, T as Trajectory } from './trajectory-2TkpSEVh.js';
|
|
51
51
|
export { b as buildTrajectory } from './trajectory-2TkpSEVh.js';
|
|
52
|
+
import { AgentProfile } from '@tangle-network/agent-interface';
|
|
53
|
+
export { AgentProfile } from '@tangle-network/agent-interface';
|
|
52
54
|
import { b as ChannelRollup, C as CostLedger } from './cost-ledger-DuSqlw5B.js';
|
|
53
55
|
export { a as CostChannel, c as CostLedgerEntry, d as CostLedgerSummary, e as CostResult, f as CostUsage, g as costForUsage, m as modelPriceKey } from './cost-ledger-DuSqlw5B.js';
|
|
54
56
|
export { D as Direction, O as Objective, P as ParetoResult, c as crowdingDistance, d as dominates, p as paretoFrontier, a as paretoFrontierWithCrowding, s as scalarScore } from './pareto-E-pembql.js';
|
|
@@ -65,15 +67,16 @@ import { b as Layer, S as Severity, L as LayerResult, c as VerifyContext } from
|
|
|
65
67
|
export { F as Finding, d as LayerStatus, M as MultiLayerVerifier, a as VerificationReport, V as VerifyOptions, g as gradeSemanticStatus } from './multi-layer-verifier-DUZXrPDA.js';
|
|
66
68
|
import { L as LlmClientOptions } from './llm-client-Bj7g0rqu.js';
|
|
67
69
|
export { d as LlmCallError, b as LlmCallRequest, c as LlmCallResult, e as LlmClient, f as LlmMessage, g as LlmRouteAssertionError, a as LlmRouteRequirements, h as LlmUsage, i as assertLlmRoute, j as backoffMs, k as callLlm, l as callLlmJson, m as isTransientLlmError, p as probeLlm, s as stripFencedJson } from './llm-client-Bj7g0rqu.js';
|
|
68
|
-
export { B as BENCHMARK_SPLIT_SEED, a as BenchmarkAdapter, b as BenchmarkDatasetItem, c as BenchmarkEvaluation, d as benchmarkDeterministicSplit, i as benchmarks } from './index-
|
|
69
|
-
export { C as CallbackResearcher, d as CallbackResearcherOptions, e as CampaignFactoryParams, f as CampaignIntegrityPolicy, g as CampaignRunContext, h as CampaignRunOutcome, i as CampaignRunner, j as CampaignScenario, k as CampaignVariant, c as EvalCampaignOptions, b as EvalCampaignResult, E as ExperimentPlan, a as ExperimentResult, l as FailedRun, F as FailureMode, N as NoopResearcher, R as Researcher, S as SteeringChange, r as runEvalCampaign } from './researcher-
|
|
70
|
-
export { G as GainDistributionBin, a as GainDistributionFigureSpec, b as GainDistributionOptions, m as GateDecision, n as GateEvidence, H as HeldOutGate, o as HeldOutGateConfig, q as HeldOutGateRejectionCode, P as ParetoFigureSpec, c as ParetoPoint, R as RESEARCH_REPORT_HARD_PAIR_FLOOR, d as ResearchReport, e as ResearchReportCandidate, f as ResearchReportDecision, g as ResearchReportMethodology, h as ResearchReportOptions, i as ResearchReportRecommendation, S as SummaryTable, j as SummaryTableOptions, k as SummaryTableRow, l as gainHistogram, p as paretoChart, r as researchReport, s as summaryTable } from './summary-report-
|
|
70
|
+
export { B as BENCHMARK_SPLIT_SEED, a as BenchmarkAdapter, b as BenchmarkDatasetItem, c as BenchmarkEvaluation, d as benchmarkDeterministicSplit, i as benchmarks } from './index-_Y4oNOOb.js';
|
|
71
|
+
export { C as CallbackResearcher, d as CallbackResearcherOptions, e as CampaignFactoryParams, f as CampaignIntegrityPolicy, g as CampaignRunContext, h as CampaignRunOutcome, i as CampaignRunner, j as CampaignScenario, k as CampaignVariant, c as EvalCampaignOptions, b as EvalCampaignResult, E as ExperimentPlan, a as ExperimentResult, l as FailedRun, F as FailureMode, N as NoopResearcher, R as Researcher, S as SteeringChange, r as runEvalCampaign } from './researcher-Jr8ME1dZ.js';
|
|
72
|
+
export { G as GainDistributionBin, a as GainDistributionFigureSpec, b as GainDistributionOptions, m as GateDecision, n as GateEvidence, H as HeldOutGate, o as HeldOutGateConfig, q as HeldOutGateRejectionCode, P as ParetoFigureSpec, c as ParetoPoint, R as RESEARCH_REPORT_HARD_PAIR_FLOOR, d as ResearchReport, e as ResearchReportCandidate, f as ResearchReportDecision, g as ResearchReportMethodology, h as ResearchReportOptions, i as ResearchReportRecommendation, S as SummaryTable, j as SummaryTableOptions, k as SummaryTableRow, l as gainHistogram, p as paretoChart, r as researchReport, s as summaryTable } from './summary-report-CInXwsza.js';
|
|
73
|
+
export { L as LockedJsonlAppender } from './testing-C21CHsq2.js';
|
|
71
74
|
export { I as InterimReleaseConfidence, a as InterimReleaseConfidenceInput, P as PairedEvalueOptions, b as PairedEvalueSequence, c as PairedEvalueStep, S as SequentialDecision, e as evaluateInterimReleaseConfidence, p as pairedEvalueSequence } from './sequential-5iSVfzl2.js';
|
|
72
|
-
import { e as
|
|
75
|
+
import { e as GepaProposerConstraints, a as RunImprovementLoopResult } from './gepa-C1NCIZ9o.js';
|
|
73
76
|
export { IntegrityResult, IntegrityViolation, JourneySpec, PerfBaseline, PerfGateResult, PerfRegression, PerfScenario, PerfStat, ScenarioAxes, assertRecordIntegrity, checkRecordIntegrity, expandMatrix, gatePerf, scenarioKey, summarizeRecords } from './perf/index.js';
|
|
74
77
|
import '@ax-llm/ax';
|
|
75
78
|
import 'zod';
|
|
76
|
-
import './insight-report-
|
|
79
|
+
import './insight-report-BnRjTibG.js';
|
|
77
80
|
import './outcome-store-rnXLEqSn.js';
|
|
78
81
|
|
|
79
82
|
/**
|
|
@@ -98,8 +101,6 @@ import './outcome-store-rnXLEqSn.js';
|
|
|
98
101
|
*
|
|
99
102
|
* Both implement the small `AutoPrClient` interface, so tests substitute
|
|
100
103
|
* a fake without spinning a process or network.
|
|
101
|
-
*
|
|
102
|
-
* @experimental — surface may evolve as consumers wire it into CI workflows.
|
|
103
104
|
*/
|
|
104
105
|
interface FileChange {
|
|
105
106
|
/** Repo-relative path. Forward slashes; no `..`. */
|
|
@@ -1652,6 +1653,31 @@ declare class BudgetGuard {
|
|
|
1652
1653
|
get state(): Record<keyof BudgetSpec, number>;
|
|
1653
1654
|
}
|
|
1654
1655
|
|
|
1656
|
+
/**
|
|
1657
|
+
* Collision-resistant, path-safe, human-readable profile id for eval artifacts.
|
|
1658
|
+
* Scorecard joins still use `agentProfileHash`; this id is for run ids, matrix
|
|
1659
|
+
* keys, and directory names where two profiles must not collapse onto one row.
|
|
1660
|
+
* The suffix is the first 64 bits of the behaviour hash, enough for ordinary
|
|
1661
|
+
* eval matrices while keeping filenames readable.
|
|
1662
|
+
*/
|
|
1663
|
+
declare function agentProfileId(profile: AgentProfile): string;
|
|
1664
|
+
/**
|
|
1665
|
+
* Model snapshot used for `RunRecord.model`. Eval surfaces require a concrete
|
|
1666
|
+
* model id because run records reject bare/missing model aliases.
|
|
1667
|
+
*/
|
|
1668
|
+
declare function agentProfileModelId(profile: AgentProfile): string;
|
|
1669
|
+
/**
|
|
1670
|
+
* Deterministic behaviour identity for the canonical
|
|
1671
|
+
* `@tangle-network/agent-interface` AgentProfile.
|
|
1672
|
+
*
|
|
1673
|
+
* `name` and `description` are labels and do not affect the hash. Profile
|
|
1674
|
+
* `version`, prompt, model hints, tools, resources, hooks, modes, permissions,
|
|
1675
|
+
* and extensions do affect the hash. Resource array order is hash-bearing
|
|
1676
|
+
* because mount order can change agent behaviour. Undefined fields are treated
|
|
1677
|
+
* as absent; explicit `null` fields remain hash-bearing.
|
|
1678
|
+
*/
|
|
1679
|
+
declare function agentProfileHash(profile: AgentProfile): string;
|
|
1680
|
+
|
|
1655
1681
|
/**
|
|
1656
1682
|
* Cost tracker — token + USD accounting per scenario and per run.
|
|
1657
1683
|
*
|
|
@@ -2283,19 +2309,19 @@ interface ScorecardCell {
|
|
|
2283
2309
|
interface Scorecard {
|
|
2284
2310
|
cells: ScorecardCell[];
|
|
2285
2311
|
/** Profile definitions seen — keeps the scorecard self-describing. */
|
|
2286
|
-
profiles: Record<string, AgentProfile
|
|
2312
|
+
profiles: Record<string, AgentProfile>;
|
|
2287
2313
|
}
|
|
2288
2314
|
/** One append-only log line — a single cell's entry for a single commit. */
|
|
2289
2315
|
interface ScorecardLogLine {
|
|
2290
2316
|
scenarioId: string;
|
|
2291
2317
|
profileHash: string;
|
|
2292
2318
|
model: string;
|
|
2293
|
-
profile: AgentProfile
|
|
2319
|
+
profile: AgentProfile;
|
|
2294
2320
|
entry: ScorecardEntry;
|
|
2295
2321
|
}
|
|
2296
2322
|
interface RecordRunsOptions {
|
|
2297
2323
|
/** The profile that produced these runs — keys the cell. */
|
|
2298
|
-
profile: AgentProfile
|
|
2324
|
+
profile: AgentProfile;
|
|
2299
2325
|
commitSha: string;
|
|
2300
2326
|
/** Defaults to `new Date().toISOString()`. */
|
|
2301
2327
|
timestamp?: string;
|
|
@@ -4712,25 +4738,6 @@ declare function precision<T>(goldens: GoldenSpec[], candidates: T[], options?:
|
|
|
4712
4738
|
text?: (candidate: T) => string;
|
|
4713
4739
|
}): number;
|
|
4714
4740
|
|
|
4715
|
-
/**
|
|
4716
|
-
* LockedJsonlAppender — mutex-serialized JSONL append helper for arbitrary
|
|
4717
|
-
* payloads. The reference-replay store does the same thing for typed
|
|
4718
|
-
* `ReferenceReplayRun` rows; this is the generic version used by
|
|
4719
|
-
* `MutationTelemetry`, `TrialTelemetry`, and any other consumer that wants
|
|
4720
|
-
* append-only durable telemetry without rolling its own lock.
|
|
4721
|
-
*
|
|
4722
|
-
* Locks are per absolute file path (process-local). Cross-process
|
|
4723
|
-
* concurrency is NOT addressed — that's an fcntl/flock problem.
|
|
4724
|
-
*/
|
|
4725
|
-
declare class LockedJsonlAppender {
|
|
4726
|
-
readonly path: string;
|
|
4727
|
-
private readonly mutex;
|
|
4728
|
-
constructor(path: string);
|
|
4729
|
-
append(entry: unknown): Promise<void>;
|
|
4730
|
-
}
|
|
4731
|
-
/** Reset all internal mutex state — tests only. */
|
|
4732
|
-
declare function resetLockedAppendersForTesting(): void;
|
|
4733
|
-
|
|
4734
4741
|
/**
|
|
4735
4742
|
* Inter-critic / inter-pass orthogonality.
|
|
4736
4743
|
*
|
|
@@ -5002,8 +5009,6 @@ declare function traceJudge(judge: JudgeFn, judgeName: string, opts: TracedJudge
|
|
|
5002
5009
|
declare function traceJudgeEnsemble(judges: JudgeFn[], judgeNames: string[], opts: TracedJudgeOptions): JudgeFn;
|
|
5003
5010
|
|
|
5004
5011
|
/**
|
|
5005
|
-
* @experimental
|
|
5006
|
-
*
|
|
5007
5012
|
* Gold scenarios for teacher→student distillation. The TEACHER is an
|
|
5008
5013
|
* expensive workflow (e.g. the 70-agent skill audit) whose verdicts are
|
|
5009
5014
|
* frozen as gold labels; the STUDENT is a cheap single-shot analyst whose
|
|
@@ -5045,7 +5050,7 @@ interface SplitGoldOptions {
|
|
|
5045
5050
|
testEveryNth?: number;
|
|
5046
5051
|
}
|
|
5047
5052
|
interface GoldSplit<TInput, TLabel> {
|
|
5048
|
-
/** Training scenarios — the optimization pool the
|
|
5053
|
+
/** Training scenarios — the optimization pool the proposer searches over. */
|
|
5049
5054
|
train: GoldScenario<TInput, TLabel>[];
|
|
5050
5055
|
/** Held-out scenarios — kept OUT of training; scored only at the gate. */
|
|
5051
5056
|
test: GoldScenario<TInput, TLabel>[];
|
|
@@ -5058,8 +5063,6 @@ interface GoldSplit<TInput, TLabel> {
|
|
|
5058
5063
|
declare function splitGold<TInput, TLabel>(scenarios: GoldScenario<TInput, TLabel>[], options?: SplitGoldOptions): GoldSplit<TInput, TLabel>;
|
|
5059
5064
|
|
|
5060
5065
|
/**
|
|
5061
|
-
* @experimental
|
|
5062
|
-
*
|
|
5063
5066
|
* Agreement judge for teacher→student distillation. Scores a STUDENT artifact
|
|
5064
5067
|
* (the cheap analyst's produced label) against the GoldScenario's gold label
|
|
5065
5068
|
* (the teacher's verdict). The score IS the distillation objective: 1.0 means
|
|
@@ -5076,13 +5079,13 @@ declare function splitGold<TInput, TLabel>(scenarios: GoldScenario<TInput, TLabe
|
|
|
5076
5079
|
*/
|
|
5077
5080
|
|
|
5078
5081
|
/** What an injected comparator returns: a [0,1] composite plus the per-field
|
|
5079
|
-
* (per-dimension) agreement breakdown the GEPA
|
|
5082
|
+
* (per-dimension) agreement breakdown the GEPA proposer reflects on to learn
|
|
5080
5083
|
* WHICH part of the verdict the student is getting wrong. */
|
|
5081
5084
|
interface AgreementResult {
|
|
5082
5085
|
/** Overall agreement in [0,1]. */
|
|
5083
5086
|
score: number;
|
|
5084
5087
|
/** Per-dimension agreement in [0,1] — keyed by field/aspect name. The
|
|
5085
|
-
* reflective
|
|
5088
|
+
* reflective proposer surfaces the weakest of these as the lever to fix. */
|
|
5086
5089
|
dimensions: Record<string, number>;
|
|
5087
5090
|
}
|
|
5088
5091
|
/** Compare a produced label against a gold label → agreement. Injected so the
|
|
@@ -5128,12 +5131,10 @@ interface FieldAgreementSpec {
|
|
|
5128
5131
|
declare function fieldAgreement<TProduced extends Record<string, unknown>, TLabel>(spec: FieldAgreementSpec): CompareLabels<TProduced, TLabel>;
|
|
5129
5132
|
|
|
5130
5133
|
/**
|
|
5131
|
-
* @experimental
|
|
5132
|
-
*
|
|
5133
5134
|
* `runDistillation` — the teacher→student distillation loop. COMPOSES existing
|
|
5134
5135
|
* substrate primitives; reimplements none of them:
|
|
5135
5136
|
*
|
|
5136
|
-
* - DRIVER = `
|
|
5137
|
+
* - DRIVER = `gepaProposer` (reflective prompt optimizer)
|
|
5137
5138
|
* - LOOP = `runImprovementLoop` (outer: optimize → holdout re-score → gate)
|
|
5138
5139
|
* - MEASUREMENT = `runCampaign` (inside the loop) scoring the student
|
|
5139
5140
|
* - JUDGE = `buildAgreementJudge` — student label vs gold teacher label
|
|
@@ -5175,7 +5176,7 @@ interface RunDistillationOptions<TProduced, TInput, TLabel> {
|
|
|
5175
5176
|
/** Transport for BOTH the student (cheap model) and the GEPA reflection
|
|
5176
5177
|
* (the optimizer model). The student calls it via `createChatClient`. */
|
|
5177
5178
|
llm: CreateChatClientOpts;
|
|
5178
|
-
/** Router transport the GEPA
|
|
5179
|
+
/** Router transport the GEPA proposer reflects through. `gepaProposer` uses the
|
|
5179
5180
|
* package `LlmClient` directly (`LlmClientOptions`), not the ChatClient —
|
|
5180
5181
|
* pass the router creds here. A test may inject `fetch` to stub the
|
|
5181
5182
|
* reflection HTTP and exercise the wiring without real tokens. */
|
|
@@ -5202,7 +5203,7 @@ interface RunDistillationOptions<TProduced, TInput, TLabel> {
|
|
|
5202
5203
|
/** Levers offered to the GEPA reflection prompt. */
|
|
5203
5204
|
mutationPrimitives?: string[];
|
|
5204
5205
|
/** GEPA structured-doc constraints (preserve sections, edit budget). */
|
|
5205
|
-
constraints?:
|
|
5206
|
+
constraints?: GepaProposerConstraints;
|
|
5206
5207
|
/** Gate's minimum holdout-agreement delta to ship. Default 0.0 — a
|
|
5207
5208
|
* distillation run reports the lift; the caller decides the bar. Only used
|
|
5208
5209
|
* when `gate` is omitted (the default `heldOutGate`). */
|
|
@@ -5243,8 +5244,6 @@ declare function defaultRenderStudentPrompt<TInput>(args: {
|
|
|
5243
5244
|
declare function defaultParseStudentLabel<TProduced>(rawContent: string, scenarioId: string): TProduced;
|
|
5244
5245
|
|
|
5245
5246
|
/**
|
|
5246
|
-
* @experimental
|
|
5247
|
-
*
|
|
5248
5247
|
* Strong, generically-useful baseline ROLES — the top zone of an `AgentProfile`
|
|
5249
5248
|
* before any domain layer. A product composes one of these with its own
|
|
5250
5249
|
* environment description (its sandbox) and its domain guidance, which lives in
|
|
@@ -5282,9 +5281,7 @@ declare const BASELINE_ROLES: {
|
|
|
5282
5281
|
type BaselineRoleKey = keyof typeof BASELINE_ROLES;
|
|
5283
5282
|
|
|
5284
5283
|
/**
|
|
5285
|
-
*
|
|
5286
|
-
*
|
|
5287
|
-
* Structured agent profile — the system prompt as named, addressable sections
|
|
5284
|
+
* Structured prompt profile — the system prompt as named, addressable sections
|
|
5288
5285
|
* instead of one opaque blob. The self-improvement loop targets ONE evolvable
|
|
5289
5286
|
* `domain` section at a time (via `applyDomainPatch`); the role, environment,
|
|
5290
5287
|
* tool conventions, and skill roster stay fixed so a candidate diff is
|
|
@@ -5297,11 +5294,9 @@ type BaselineRoleKey = keyof typeof BASELINE_ROLES;
|
|
|
5297
5294
|
* loop's string `MutableSurface`: a profile renders to exactly the text a
|
|
5298
5295
|
* candidate is scored on.
|
|
5299
5296
|
*
|
|
5300
|
-
* Distinct from the
|
|
5301
|
-
*
|
|
5302
|
-
*
|
|
5303
|
-
* varied. Consume this module by its own path (`@tangle-network/agent-eval`
|
|
5304
|
-
* exposes it under the `profile` namespace) to avoid the name clash.
|
|
5297
|
+
* Distinct from the canonical `AgentProfile` exported by
|
|
5298
|
+
* `@tangle-network/agent-interface`. This helper builds prompt content that can
|
|
5299
|
+
* be rendered into an agent-interface profile's `prompt.systemPrompt`.
|
|
5305
5300
|
*/
|
|
5306
5301
|
|
|
5307
5302
|
/** A named, addressable region of the system prompt. `evolvable` marks whether
|
|
@@ -5324,7 +5319,7 @@ interface ProfileSkill {
|
|
|
5324
5319
|
}
|
|
5325
5320
|
/** The structured system prompt. The first four fields are fixed scaffolding;
|
|
5326
5321
|
* `domain` is the evolvable surface the loop optimizes. */
|
|
5327
|
-
interface
|
|
5322
|
+
interface PromptProfile {
|
|
5328
5323
|
role: string;
|
|
5329
5324
|
environment: string;
|
|
5330
5325
|
toolConventions: string;
|
|
@@ -5338,10 +5333,10 @@ interface AgentProfile {
|
|
|
5338
5333
|
* `### <title>` + body). The order is load-bearing — the loop diffs rendered
|
|
5339
5334
|
* text, and reordering would make every candidate look like a full rewrite.
|
|
5340
5335
|
*/
|
|
5341
|
-
declare function renderProfile(p:
|
|
5336
|
+
declare function renderProfile(p: PromptProfile): string;
|
|
5342
5337
|
/** The string `MutableSurface` the self-improvement loop scores — a profile
|
|
5343
5338
|
* renders to exactly the text a candidate is graded on. */
|
|
5344
|
-
declare function profileToSurface(p:
|
|
5339
|
+
declare function profileToSurface(p: PromptProfile): string;
|
|
5345
5340
|
/**
|
|
5346
5341
|
* The fixed-scaffolding baseline: a `role`/`environment`/`toolConventions`
|
|
5347
5342
|
* profile with the stock skill roster and NO domain guidance. The
|
|
@@ -5352,7 +5347,7 @@ declare function baselineProfile(args: {
|
|
|
5352
5347
|
environment?: string;
|
|
5353
5348
|
toolConventions?: string;
|
|
5354
5349
|
skills?: ProfileSkill[];
|
|
5355
|
-
}):
|
|
5350
|
+
}): PromptProfile;
|
|
5356
5351
|
/**
|
|
5357
5352
|
* Baseline from one of the strong generic roles (`'engineer' | 'researcher' |
|
|
5358
5353
|
* 'generalist'`) — the common case: pick a role foundation, optionally override
|
|
@@ -5364,12 +5359,12 @@ declare function baselineProfileFromRole(role: BaselineRoleKey, args?: {
|
|
|
5364
5359
|
environment?: string;
|
|
5365
5360
|
toolConventions?: string;
|
|
5366
5361
|
skills?: ProfileSkill[];
|
|
5367
|
-
}):
|
|
5362
|
+
}): PromptProfile;
|
|
5368
5363
|
/** The production profile: the baseline scaffolding plus the domain sections
|
|
5369
5364
|
* shipped after self-improvement. Differs from the baseline ONLY in `domain`
|
|
5370
5365
|
* (and any skills the caller layered into the baseline) — the role,
|
|
5371
5366
|
* environment, and tool conventions are carried through unchanged. */
|
|
5372
|
-
declare function prodProfile(baseline:
|
|
5367
|
+
declare function prodProfile(baseline: PromptProfile, shipped: AgentProfileSection[]): PromptProfile;
|
|
5373
5368
|
/**
|
|
5374
5369
|
* Section-scoped edit — replace the body of ONE domain section by id, leaving
|
|
5375
5370
|
* every other section byte-identical. This is how the loop targets a single
|
|
@@ -5377,7 +5372,7 @@ declare function prodProfile(baseline: AgentProfile, shipped: AgentProfileSectio
|
|
|
5377
5372
|
* section is a silent no-op otherwise — fail loud). Throws when the targeted
|
|
5378
5373
|
* section is `evolvable: false`.
|
|
5379
5374
|
*/
|
|
5380
|
-
declare function applyDomainPatch(p:
|
|
5375
|
+
declare function applyDomainPatch(p: PromptProfile, sectionId: string, newBody: string): PromptProfile;
|
|
5381
5376
|
/**
|
|
5382
5377
|
* Content hash of a single section — its loop identity. Delegates to the
|
|
5383
5378
|
* campaign provenance `surfaceContentHash` (the same helper the loop's
|
|
@@ -5389,11 +5384,11 @@ declare function applyDomainPatch(p: AgentProfile, sectionId: string, newBody: s
|
|
|
5389
5384
|
*/
|
|
5390
5385
|
declare function sectionHash(section: AgentProfileSection): string;
|
|
5391
5386
|
|
|
5392
|
-
type index_AgentProfile = AgentProfile;
|
|
5393
5387
|
type index_AgentProfileSection = AgentProfileSection;
|
|
5394
5388
|
declare const index_BASELINE_ROLES: typeof BASELINE_ROLES;
|
|
5395
5389
|
type index_BaselineRoleKey = BaselineRoleKey;
|
|
5396
5390
|
type index_ProfileSkill = ProfileSkill;
|
|
5391
|
+
type index_PromptProfile = PromptProfile;
|
|
5397
5392
|
declare const index_applyDomainPatch: typeof applyDomainPatch;
|
|
5398
5393
|
declare const index_baselineProfile: typeof baselineProfile;
|
|
5399
5394
|
declare const index_baselineProfileFromRole: typeof baselineProfileFromRole;
|
|
@@ -5405,7 +5400,7 @@ declare const index_renderProfile: typeof renderProfile;
|
|
|
5405
5400
|
declare const index_researcherRole: typeof researcherRole;
|
|
5406
5401
|
declare const index_sectionHash: typeof sectionHash;
|
|
5407
5402
|
declare namespace index {
|
|
5408
|
-
export { type
|
|
5403
|
+
export { type index_AgentProfileSection as AgentProfileSection, index_BASELINE_ROLES as BASELINE_ROLES, type index_BaselineRoleKey as BaselineRoleKey, type index_ProfileSkill as ProfileSkill, type index_PromptProfile as PromptProfile, index_applyDomainPatch as applyDomainPatch, index_baselineProfile as baselineProfile, index_baselineProfileFromRole as baselineProfileFromRole, index_engineerRole as engineerRole, index_generalistRole as generalistRole, index_prodProfile as prodProfile, index_profileToSurface as profileToSurface, index_renderProfile as renderProfile, index_researcherRole as researcherRole, index_sectionHash as sectionHash };
|
|
5409
5404
|
}
|
|
5410
5405
|
|
|
5411
5406
|
/**
|
|
@@ -5465,8 +5460,8 @@ declare function attachCostToReport<R extends object>(report: R, ledger: CostLed
|
|
|
5465
5460
|
* and the `JudgeConfig`s handed to `makeEvalTools({ judges })`
|
|
5466
5461
|
* (src/eval-tools.ts).
|
|
5467
5462
|
* - `reflection` → `selfImprove({ llm: { model: seats.reflection } })` — the
|
|
5468
|
-
* `
|
|
5469
|
-
* same seat for any custom `
|
|
5463
|
+
* `gepaProposer` reflection model (src/contract/self-improve.ts);
|
|
5464
|
+
* same seat for any custom `SurfaceProposer`'s LLM.
|
|
5470
5465
|
* - `worker` → the dispatch model the agent itself calls — the model an
|
|
5471
5466
|
* `AgentProfile` declares.
|
|
5472
5467
|
* - `analyst` → the LLM behind `analyzeRuns` / analyst-registry kinds.
|
|
@@ -5485,7 +5480,7 @@ interface ModelSeats {
|
|
|
5485
5480
|
judges?: string[];
|
|
5486
5481
|
/** Analyst model — `analyzeRuns` / analyst-registry LLM calls. */
|
|
5487
5482
|
analyst?: string;
|
|
5488
|
-
/** Reflection/
|
|
5483
|
+
/** Reflection/proposer model — `gepaProposer` mutation proposals. */
|
|
5489
5484
|
reflection?: string;
|
|
5490
5485
|
/** Verifier model — completion/objective checking. */
|
|
5491
5486
|
verifier?: string;
|
|
@@ -5663,4 +5658,4 @@ type CachedJudge<TArtifact, TScenario extends Scenario$1 = Scenario$1> = JudgeCo
|
|
|
5663
5658
|
*/
|
|
5664
5659
|
declare function cachedJudge<TArtifact, TScenario extends Scenario$1 = Scenario$1>(judge: JudgeConfig<TArtifact, TScenario>, store: VerdictCacheStore, options: CachedJudgeOptions): CachedJudge<TArtifact, TScenario>;
|
|
5665
5660
|
|
|
5666
|
-
export { ATTESTATION_ALGORITHM, type ActiveLearningOptions, type AdapterRun, AgentDriver, type AgentDriverConfig, AgentEvalError, AgentProfile$1 as AgentProfile, type AgreementResult, type AlignmentOp, AnalyzeTracesInput, AnalyzeTracesOptions, AnalyzeTracesResult, type AntiSlopConfig, type AntiSlopIssue, type AntiSlopReport, type AssertCrossFamilyOptions, type AssertSingleBackendOptions, type AttestationProvenance, type AttestationVerification, type AttestedReport, type AutoPrClient, AxGepaSteeringOptimizer, type AxSteeringOptimizerConfig, type BackendDescriptor, BaselineReport, BehaviorAssertion, BenchmarkReport, BenchmarkRunner, BenchmarkRunnerConfig, type BisectOptions, type BisectResult, type BisectStep, BudgetBreachError, BudgetGuard, BudgetLedgerEntry, BudgetSpec, type BuildAgreementJudgeOptions, type CachedJudge, type CachedJudgeOptions, CallExpectation, type CanaryAlert, type CanaryKind, type CanaryLeak, type CanaryOptions, type CanaryReport, type CanarySeverity, type CandidateComparison, type CandidateScenario, type CausalAttributionReport, type CellVerdict, ChannelRollup, ChatRequest, CheckResult, CollectedArtifacts, type CommandRunner, type CompareLabels, CompletionCriterion, ConfigError, type ContinuityCheck, type ContinuityCheckResult, type ContinuityReport, type ContinuitySnapshotPair, type ContractCheckResult, type ContractJudgeOptions, type ContractMetric, type ContractReport, type ContractRule, type ContractRuleKind, type ContractSpan, type ContractVerdict, type ContractViolation, ConvergenceTracker, CorrectnessChecker, type CostEntry, CostLedger, type CostReport, type CostSummary, CostTracker, CreateChatClientOpts, type CreateDefaultReviewerOptions, type CreateExperimentInput, type CreateSandboxPoolOpts, CrossFamilyError, type CrossTraceDiff, type CrossTraceDiffOptions, DEFAULT_AGENT_SLOS, DEFAULT_FINDERS, DEFAULT_MUTATION_PRIMITIVES, DEFAULT_MUTATORS, DEFAULT_PR_REVIEW_SCORE_WEIGHTS, DEFAULT_SEVERITY_WEIGHTS, Dataset, DatasetScenario, type DecideNextUserTurnOpts, DefaultVerdict, type DeployFamily, type DeployGateLayerInput, type DeployRunResult, type DeployRunner, type DescriptionLengthCandidate, type DescriptionLengthConfig, type DescriptionLengthDecision, type DescriptionLengthEvidence, DescriptionLengthGate, type DescriptionLengthRejectionCode, type DetectorEvent, type DetectorSeverity, type DetectorSignal, type DiffScorecardOptions, type DirEntry, type DiscoverPersonasOptions, type DiscoveredPersona, DriverResult, DriverState, DualAgentBench, type DualAgentBenchConfig, type DualAgentReport, type DualAgentRound, type DualAgentScenario, type DualAgentScenarioResult, ERROR_COUNT_PATTERNS, type EnsembleAggregate, type EnsembleJudgeOptions, type ErrorCountPattern, type ErrorStreakOptions, type EvalToolDef, EvalTraceStore, type EvolutionRound, type ExecutorConfig, type Expectation, type Experiment, type ExperimentProvenance, type ExperimentRep, type ExperimentStats, type ExperimentStore, ExperimentTracker, type ExperimentTrackerOptions, type ExperimentVerdict, type ExportedRewardModel, type ExtractOptions, type ExtractResult, type FactorContribution, type FactorialCell, FailureClass, FeedbackLabel, FeedbackTrajectory, FeedbackTrajectoryStore, type FieldAgreementSpec, type FileChange, type FlowAction, type FlowLayerEnv, type FlowLayerFactoryInput, type FlowRunner, type FlowRunnerStepResult, type FlowSpec, type FlowStep, type GhCliClientOptions, type GoldScenario, type GoldSplit, type GoldenSeverity, type GoldenSpec, HarnessConfig, type HeldOutPartition, HoldoutAuditor, type HttpGithubClientOptions, INTENT_MATCH_JUDGE_VERSION, type ImageData, type ImprovementThresholds, type ImprovementVerdictResult, InMemoryWorkspaceInspector, type InferenceScorer, type InspectorContext, type IntentMatchInput, type IntentMatchOptions, type IntentMatchResult, type InteractionContribution, JudgeError, type JudgeFamily, type JudgeFleetOptions, JudgeFn, JudgeParseError, type JudgeReplayResult, type JudgeRetryOutcome, type JudgeRetryPolicy, JudgeRunner, type JudgeVerdict, type KeywordConceptSpec, type KeywordCoverageFinding, type KeywordCoverageOptions, type KeywordCoverageResult, type LangfuseEnvelope, type LangfuseGeneration, type LangfuseScore, Layer, LayerResult, type LiveProofArtifact, type LiveProofConfig, type LiveProofContext, type LiveProofResult, LlmClientOptions, LlmSpan, LockedJsonlAppender, MODEL_PRICING, type MakeEvalToolsConfig, type MatchResult, type MatcherResult, type MergeOptions, MetricsCollector, type ModelCostRollup, type ModelPreflight, type ModelSeats, ModelsUnreachableError, type MuffledFinder, type MuffledFinding, type MultiToolchainLayerConfig, type Mutator, Mutex, type NoProgressOptions, type Oracle, type OracleObservation, type OracleReport, type OracleResult, type OrthogonalityInput, type OrthogonalityResult, OtelExportConfig, OtelExporter, type OtelPipelineHandle, type OtelPipelineOptions, PairwiseSteeringOptimizer, type ParaphraseRobustnessScenarioInput, type ParaphraseRobustnessScenarioResult, type ParseStudentLabel, type PartitionHeldOutOptions, PersonaConfig, type Playbook, type PlaybookEntry, type PoolSlot, type PrReviewAuditCase, type PrReviewBenchmarkSummary, type PrReviewComment, type PrReviewMatchedFinding, type PrReviewOutcome, type PrReviewReferenceFinding, type PrReviewScore, type PrReviewScoreWeights, type PrReviewSeverity, type PrReviewSource, type PreflightModelsOptions, type PreflightOutcome, ProductClient, ProductClientConfig, type PromptHandle, PromptRegistry, type ProposeAutomatedPullRequestInput, type ProposeAutomatedPullRequestResult, type ProvenanceReader, type RecordRunsOptions, type ReferenceMatchResult, type ReferenceReplayAdapter, type ReferenceReplayAdapterFn, type ReferenceReplayAdapterLike, type ReferenceReplayAggregate, type ReferenceReplayCandidate, type ReferenceReplayCase, type ReferenceReplayCaseRun, type ReferenceReplayExecutionScenario, type ReferenceReplayItem, type ReferenceReplayMatch, type ReferenceReplayMatchStrategy, type ReferenceReplayMatcher, type ReferenceReplayPromotionDecision, type ReferenceReplayPromotionPolicy, type ReferenceReplayRun, type ReferenceReplayRunContext, type ReferenceReplayRunOptions, type ReferenceReplayRunStore, type ReferenceReplayScenario, type ReferenceReplayScenarioScore, type ReferenceReplayScore, type ReferenceReplayScoreOptions, type ReferenceReplaySplit, type ReferenceReplaySplitComparison, type ReferenceReplaySteeringRowsOptions, type ReflectionContext, type ReflectionProposal, ReleaseConfidenceScorecard, ReleaseConfidenceThresholds, type RenderStudentPrompt, type RepeatedActionOptions, type RepoRef, type ReviewerMemoryEntry, type ReviewerOutput, type ReviewerPromptInput, type ReviewerSoftFailDefaults, type ReviewerVerificationSummary, type RobustnessResult, Run, type RunCommandInput, type RunCommandResult, type RunDistillationOptions, type RunDistillationResult, RunFilter, RunRecord, type RunRecordBackend, type RunRecordFilter, RunScore, RunScoreWeights, RunSplitTag, SandboxDriver, SandboxHarnessResult, type SandboxJudgeKind, type SandboxJudgeResult, type SandboxJudgeSpec, type SandboxPool, type ScanOptions, Scenario, type ScenarioCost, ScenarioFile, ScenarioRegistry, ScenarioResult, type Scorecard, type ScorecardCell, type ScorecardCellDiff, type ScorecardDiff, type ScorecardEntry, type ScorecardLogLine, type ScoredTarget, type SeatName, type SeatPresetName, SeatUnsetError, type SelfPlayOptions, type SelfPlayProposer, type SelfPlayScorer, type SerializedRegex, Severity, type SingleBackendDivergence, SingleBackendError, type SingleBackendField, type SingleBackendReport, type Slo, type SloCheckResult, type SloComparator, type SloReport, type SloSeverity, type SlopCategory, type SlotFactory, type SpanPredicate, type SplitGoldOptions, SteeringBundle, type SteeringOptimizationResult, type SteeringOptimizationRow, type SteeringOptimizationSelector, type SteeringOptimizerBackend, type SteeringOptimizerConfig, type StepAttribution, type StreamingDetector, type SynthesisReason, type SynthesisTarget, TestResult, type TextMatcher, type ThresholdContract, TokenCounter, type TokenSpec, type TraceContract, TraceContractBuilder, TraceEmitter, TraceStore, type TracedAnalystOptions, type TracedJudgeOptions, Trajectory, TrajectoryStep, type TrialTrace, TurnMetrics, UI_FINDING_SEVERITIES, UI_LENSES, UNIVERSAL_FINDERS, type UiFinding, type UiFindingScreenshot, type UiFindingSeverity, type UiLens, type VerdictCacheStats, type VerdictCacheStore, VerifyContext, type VisualDiffOptions, type VisualDiffResult, type ViteDeployRunnerInput, type WorkerDriverContext, type WorkspaceAssertion, type WorkspaceAssertionResult, type WorkspaceInspector, type WorkspaceSnapshot, type WranglerDeployRunnerInput, adversarialJudge, aggregateJudgeVerdicts, aggregatePrReviewScore, analyzeAntiSlop, appendScorecard, assertCrossFamily, assertModelsServed, assertSingleBackend, assignHeldOutTag, attachCostToReport, attest, bisect, buildAgreementJudge, buildDriverSystemPrompt, buildReflectionPrompt, buildReviewerPrompt, buildWorkerDriverSystemPrompt, cachedJudge, canaryLeakView, canonicalJson, causalAttribution, checkBehavioralCanary, checkCanaries, checkSlos, checkTraceContracts, codeExecutionJudge, coherenceJudge, collectionPreserved, commentsForSource, commitBisect, compareReferenceReplay, compilerJudge, computeExperimentStats, contentHash, contractJudge, costReport, createAntiSlopJudge, createCustomJudge, createDefaultReviewer, createDomainExpertJudge, createIntentMatchJudge, createSandboxPool, crossTraceDiff, dataDescriptionBits, decideNextUserTurn, decideReferenceReplayPromotion, decideReferenceReplayRunPromotion, defaultJudges, defaultParseStudentLabel, defaultReferenceReplayMatcher, defaultRenderStudentPrompt, deployGateLayer, diffScorecard, discoverPersonas, distillPlaybook, ensembleJudge, errorStreakDetector, estimateCost, estimateTokens, evaluateContract, evaluateOracles, evaluateTraceContract, executeScenario, expectAgent, exportRewardModel, extractAssetUrls, extractErrorCount, fieldAgreement, fileContains, fileExists, fileExperimentStore, fileVerdictCache, findAutoMatchNoExpectation, findConstructorCwdDropped, findFallbackToPass, findLiteralTruePass, findSkipCountsAsPass, flowLayer, fnv1a32, formatBenchmarkReport, formatDriverReport, formatFindings, formatScorecardDiff, ghCliClient, gitProvenanceReader, precision as goldenPrecision, hashContent, hashToUnit, htmlContainsElement, httpGithubClient, improvementVerdict, inMemoryExperimentStore, inMemoryReferenceReplayStore, inMemoryRunRecordBackend, inMemoryVerdictCache, isModelPriced, isOtelConfigured, jsonShape, jsonlReferenceReplayStore, jsonlRunRecordBackend, judgeFamily, keyPreserved, linterJudge, loadGoldScenarios, loadScorecard, loadScorerFromGrader, localCommandRunner, lowercaseMutator, makeEvalTools, matchGoldens, matchSpan, mergeLayerResults, modelDescriptionBits, multiToolchainLayer, noProgressDetector, notBlocked, observeAll, paraphraseRobustness, paraphraseRobustnessScenarios, parseGoldJsonl, parseReflectionResponse, partitionHeldOut, passOrthogonality, pixelDeltaRatio, politenessPrefixMutator, preflightModels, printDriverSummary, index as profile, promptBisect, proposeAutomatedPullRequest, proposeSynthesisTargets, recordRuns, recordRunsToScorecard, referenceReplayRunsToSteeringRows, referenceReplayScenarioToRunScore, regexMatches, renderMarkdownReport, renderPlaybookMarkdown, repeatedActionDetector, replayScorerOverCorpus, replayTraceThroughJudge, resetLockedAppendersForTesting, resolveModelPricing, resolveSeat, rowCount, rowWhere, runAssertions, runBehavioralCanaries, runCanaries, runDistillation, runE2EWorkflow, runExpectations, runIntentMatchJudge, runJudgeFleet, runKeywordCoverageJudge, runKeywordCoverageJudgeUrl, runLiveProof, runReferenceReplay, runScore, runSelfPlay, scanForMuffledGates, scoreContinuity, scorePrReviewComments, scorePrReviewSource, scoreReferenceReplay, seatPresets, securityJudge, sentenceReorderMutator, splitGold, statusAdvanced, summarizePrReviewBenchmark, testJudge, textInSnapshot, toLangfuseEnvelope, toOpenAiTool, toPrometheusText, traceContract, traceJudge, traceJudgeEnsemble, tracedAnalyzeTraces, typoMutator, urlContains, verifyAttestation, visualDiff, viteDeployRunner, weightedRecall, whitespaceCollapseMutator, withJudgeRetry, withOtelPipeline, wranglerDeployRunner };
|
|
5661
|
+
export { ATTESTATION_ALGORITHM, type ActiveLearningOptions, type AdapterRun, AgentDriver, type AgentDriverConfig, AgentEvalError, type AgreementResult, type AlignmentOp, AnalyzeTracesInput, AnalyzeTracesOptions, AnalyzeTracesResult, type AntiSlopConfig, type AntiSlopIssue, type AntiSlopReport, type AssertCrossFamilyOptions, type AssertSingleBackendOptions, type AttestationProvenance, type AttestationVerification, type AttestedReport, type AutoPrClient, AxGepaSteeringOptimizer, type AxSteeringOptimizerConfig, type BackendDescriptor, BaselineReport, BehaviorAssertion, BenchmarkReport, BenchmarkRunner, BenchmarkRunnerConfig, type BisectOptions, type BisectResult, type BisectStep, BudgetBreachError, BudgetGuard, BudgetLedgerEntry, BudgetSpec, type BuildAgreementJudgeOptions, type CachedJudge, type CachedJudgeOptions, CallExpectation, type CanaryAlert, type CanaryKind, type CanaryLeak, type CanaryOptions, type CanaryReport, type CanarySeverity, type CandidateComparison, type CandidateScenario, type CausalAttributionReport, type CellVerdict, ChannelRollup, ChatRequest, CheckResult, CollectedArtifacts, type CommandRunner, type CompareLabels, CompletionCriterion, ConfigError, type ContinuityCheck, type ContinuityCheckResult, type ContinuityReport, type ContinuitySnapshotPair, type ContractCheckResult, type ContractJudgeOptions, type ContractMetric, type ContractReport, type ContractRule, type ContractRuleKind, type ContractSpan, type ContractVerdict, type ContractViolation, ConvergenceTracker, CorrectnessChecker, type CostEntry, CostLedger, type CostReport, type CostSummary, CostTracker, CreateChatClientOpts, type CreateDefaultReviewerOptions, type CreateExperimentInput, type CreateSandboxPoolOpts, CrossFamilyError, type CrossTraceDiff, type CrossTraceDiffOptions, DEFAULT_AGENT_SLOS, DEFAULT_FINDERS, DEFAULT_MUTATION_PRIMITIVES, DEFAULT_MUTATORS, DEFAULT_PR_REVIEW_SCORE_WEIGHTS, DEFAULT_SEVERITY_WEIGHTS, Dataset, DatasetScenario, type DecideNextUserTurnOpts, DefaultVerdict, type DeployFamily, type DeployGateLayerInput, type DeployRunResult, type DeployRunner, type DescriptionLengthCandidate, type DescriptionLengthConfig, type DescriptionLengthDecision, type DescriptionLengthEvidence, DescriptionLengthGate, type DescriptionLengthRejectionCode, type DetectorEvent, type DetectorSeverity, type DetectorSignal, type DiffScorecardOptions, type DirEntry, type DiscoverPersonasOptions, type DiscoveredPersona, DriverResult, DriverState, DualAgentBench, type DualAgentBenchConfig, type DualAgentReport, type DualAgentRound, type DualAgentScenario, type DualAgentScenarioResult, ERROR_COUNT_PATTERNS, type EnsembleAggregate, type EnsembleJudgeOptions, type ErrorCountPattern, type ErrorStreakOptions, type EvalToolDef, EvalTraceStore, type EvolutionRound, type ExecutorConfig, type Expectation, type Experiment, type ExperimentProvenance, type ExperimentRep, type ExperimentStats, type ExperimentStore, ExperimentTracker, type ExperimentTrackerOptions, type ExperimentVerdict, type ExportedRewardModel, type ExtractOptions, type ExtractResult, type FactorContribution, type FactorialCell, FailureClass, FeedbackLabel, FeedbackTrajectory, FeedbackTrajectoryStore, type FieldAgreementSpec, type FileChange, type FlowAction, type FlowLayerEnv, type FlowLayerFactoryInput, type FlowRunner, type FlowRunnerStepResult, type FlowSpec, type FlowStep, type GhCliClientOptions, type GoldScenario, type GoldSplit, type GoldenSeverity, type GoldenSpec, HarnessConfig, type HeldOutPartition, HoldoutAuditor, type HttpGithubClientOptions, INTENT_MATCH_JUDGE_VERSION, type ImageData, type ImprovementThresholds, type ImprovementVerdictResult, InMemoryWorkspaceInspector, type InferenceScorer, type InspectorContext, type IntentMatchInput, type IntentMatchOptions, type IntentMatchResult, type InteractionContribution, JudgeError, type JudgeFamily, type JudgeFleetOptions, JudgeFn, JudgeParseError, type JudgeReplayResult, type JudgeRetryOutcome, type JudgeRetryPolicy, JudgeRunner, type JudgeVerdict, type KeywordConceptSpec, type KeywordCoverageFinding, type KeywordCoverageOptions, type KeywordCoverageResult, type LangfuseEnvelope, type LangfuseGeneration, type LangfuseScore, Layer, LayerResult, type LiveProofArtifact, type LiveProofConfig, type LiveProofContext, type LiveProofResult, LlmClientOptions, LlmSpan, MODEL_PRICING, type MakeEvalToolsConfig, type MatchResult, type MatcherResult, type MergeOptions, MetricsCollector, type ModelCostRollup, type ModelPreflight, type ModelSeats, ModelsUnreachableError, type MuffledFinder, type MuffledFinding, type MultiToolchainLayerConfig, type Mutator, Mutex, type NoProgressOptions, type Oracle, type OracleObservation, type OracleReport, type OracleResult, type OrthogonalityInput, type OrthogonalityResult, OtelExportConfig, OtelExporter, type OtelPipelineHandle, type OtelPipelineOptions, PairwiseSteeringOptimizer, type ParaphraseRobustnessScenarioInput, type ParaphraseRobustnessScenarioResult, type ParseStudentLabel, type PartitionHeldOutOptions, PersonaConfig, type Playbook, type PlaybookEntry, type PoolSlot, type PrReviewAuditCase, type PrReviewBenchmarkSummary, type PrReviewComment, type PrReviewMatchedFinding, type PrReviewOutcome, type PrReviewReferenceFinding, type PrReviewScore, type PrReviewScoreWeights, type PrReviewSeverity, type PrReviewSource, type PreflightModelsOptions, type PreflightOutcome, ProductClient, ProductClientConfig, type PromptHandle, PromptRegistry, type ProposeAutomatedPullRequestInput, type ProposeAutomatedPullRequestResult, type ProvenanceReader, type RecordRunsOptions, type ReferenceMatchResult, type ReferenceReplayAdapter, type ReferenceReplayAdapterFn, type ReferenceReplayAdapterLike, type ReferenceReplayAggregate, type ReferenceReplayCandidate, type ReferenceReplayCase, type ReferenceReplayCaseRun, type ReferenceReplayExecutionScenario, type ReferenceReplayItem, type ReferenceReplayMatch, type ReferenceReplayMatchStrategy, type ReferenceReplayMatcher, type ReferenceReplayPromotionDecision, type ReferenceReplayPromotionPolicy, type ReferenceReplayRun, type ReferenceReplayRunContext, type ReferenceReplayRunOptions, type ReferenceReplayRunStore, type ReferenceReplayScenario, type ReferenceReplayScenarioScore, type ReferenceReplayScore, type ReferenceReplayScoreOptions, type ReferenceReplaySplit, type ReferenceReplaySplitComparison, type ReferenceReplaySteeringRowsOptions, type ReflectionContext, type ReflectionProposal, ReleaseConfidenceScorecard, ReleaseConfidenceThresholds, type RenderStudentPrompt, type RepeatedActionOptions, type RepoRef, type ReviewerMemoryEntry, type ReviewerOutput, type ReviewerPromptInput, type ReviewerSoftFailDefaults, type ReviewerVerificationSummary, type RobustnessResult, Run, type RunCommandInput, type RunCommandResult, type RunDistillationOptions, type RunDistillationResult, RunFilter, RunRecord, type RunRecordBackend, type RunRecordFilter, RunScore, RunScoreWeights, RunSplitTag, SandboxDriver, SandboxHarnessResult, type SandboxJudgeKind, type SandboxJudgeResult, type SandboxJudgeSpec, type SandboxPool, type ScanOptions, Scenario, type ScenarioCost, ScenarioFile, ScenarioRegistry, ScenarioResult, type Scorecard, type ScorecardCell, type ScorecardCellDiff, type ScorecardDiff, type ScorecardEntry, type ScorecardLogLine, type ScoredTarget, type SeatName, type SeatPresetName, SeatUnsetError, type SelfPlayOptions, type SelfPlayProposer, type SelfPlayScorer, type SerializedRegex, Severity, type SingleBackendDivergence, SingleBackendError, type SingleBackendField, type SingleBackendReport, type Slo, type SloCheckResult, type SloComparator, type SloReport, type SloSeverity, type SlopCategory, type SlotFactory, type SpanPredicate, type SplitGoldOptions, SteeringBundle, type SteeringOptimizationResult, type SteeringOptimizationRow, type SteeringOptimizationSelector, type SteeringOptimizerBackend, type SteeringOptimizerConfig, type StepAttribution, type StreamingDetector, type SynthesisReason, type SynthesisTarget, TestResult, type TextMatcher, type ThresholdContract, TokenCounter, type TokenSpec, type TraceContract, TraceContractBuilder, TraceEmitter, TraceStore, type TracedAnalystOptions, type TracedJudgeOptions, Trajectory, TrajectoryStep, type TrialTrace, TurnMetrics, UI_FINDING_SEVERITIES, UI_LENSES, UNIVERSAL_FINDERS, type UiFinding, type UiFindingScreenshot, type UiFindingSeverity, type UiLens, type VerdictCacheStats, type VerdictCacheStore, VerifyContext, type VisualDiffOptions, type VisualDiffResult, type ViteDeployRunnerInput, type WorkerDriverContext, type WorkspaceAssertion, type WorkspaceAssertionResult, type WorkspaceInspector, type WorkspaceSnapshot, type WranglerDeployRunnerInput, adversarialJudge, agentProfileHash, agentProfileId, agentProfileModelId, aggregateJudgeVerdicts, aggregatePrReviewScore, analyzeAntiSlop, appendScorecard, assertCrossFamily, assertModelsServed, assertSingleBackend, assignHeldOutTag, attachCostToReport, attest, bisect, buildAgreementJudge, buildDriverSystemPrompt, buildReflectionPrompt, buildReviewerPrompt, buildWorkerDriverSystemPrompt, cachedJudge, canaryLeakView, canonicalJson, causalAttribution, checkBehavioralCanary, checkCanaries, checkSlos, checkTraceContracts, codeExecutionJudge, coherenceJudge, collectionPreserved, commentsForSource, commitBisect, compareReferenceReplay, compilerJudge, computeExperimentStats, contentHash, contractJudge, costReport, createAntiSlopJudge, createCustomJudge, createDefaultReviewer, createDomainExpertJudge, createIntentMatchJudge, createSandboxPool, crossTraceDiff, dataDescriptionBits, decideNextUserTurn, decideReferenceReplayPromotion, decideReferenceReplayRunPromotion, defaultJudges, defaultParseStudentLabel, defaultReferenceReplayMatcher, defaultRenderStudentPrompt, deployGateLayer, diffScorecard, discoverPersonas, distillPlaybook, ensembleJudge, errorStreakDetector, estimateCost, estimateTokens, evaluateContract, evaluateOracles, evaluateTraceContract, executeScenario, expectAgent, exportRewardModel, extractAssetUrls, extractErrorCount, fieldAgreement, fileContains, fileExists, fileExperimentStore, fileVerdictCache, findAutoMatchNoExpectation, findConstructorCwdDropped, findFallbackToPass, findLiteralTruePass, findSkipCountsAsPass, flowLayer, fnv1a32, formatBenchmarkReport, formatDriverReport, formatFindings, formatScorecardDiff, ghCliClient, gitProvenanceReader, precision as goldenPrecision, hashContent, hashToUnit, htmlContainsElement, httpGithubClient, improvementVerdict, inMemoryExperimentStore, inMemoryReferenceReplayStore, inMemoryRunRecordBackend, inMemoryVerdictCache, isModelPriced, isOtelConfigured, jsonShape, jsonlReferenceReplayStore, jsonlRunRecordBackend, judgeFamily, keyPreserved, linterJudge, loadGoldScenarios, loadScorecard, loadScorerFromGrader, localCommandRunner, lowercaseMutator, makeEvalTools, matchGoldens, matchSpan, mergeLayerResults, modelDescriptionBits, multiToolchainLayer, noProgressDetector, notBlocked, observeAll, paraphraseRobustness, paraphraseRobustnessScenarios, parseGoldJsonl, parseReflectionResponse, partitionHeldOut, passOrthogonality, pixelDeltaRatio, politenessPrefixMutator, preflightModels, printDriverSummary, index as profile, promptBisect, proposeAutomatedPullRequest, proposeSynthesisTargets, recordRuns, recordRunsToScorecard, referenceReplayRunsToSteeringRows, referenceReplayScenarioToRunScore, regexMatches, renderMarkdownReport, renderPlaybookMarkdown, repeatedActionDetector, replayScorerOverCorpus, replayTraceThroughJudge, resolveModelPricing, resolveSeat, rowCount, rowWhere, runAssertions, runBehavioralCanaries, runCanaries, runDistillation, runE2EWorkflow, runExpectations, runIntentMatchJudge, runJudgeFleet, runKeywordCoverageJudge, runKeywordCoverageJudgeUrl, runLiveProof, runReferenceReplay, runScore, runSelfPlay, scanForMuffledGates, scoreContinuity, scorePrReviewComments, scorePrReviewSource, scoreReferenceReplay, seatPresets, securityJudge, sentenceReorderMutator, splitGold, statusAdvanced, summarizePrReviewBenchmark, testJudge, textInSnapshot, toLangfuseEnvelope, toOpenAiTool, toPrometheusText, traceContract, traceJudge, traceJudgeEnsemble, tracedAnalyzeTraces, typoMutator, urlContains, verifyAttestation, visualDiff, viteDeployRunner, weightedRecall, whitespaceCollapseMutator, withJudgeRetry, withOtelPipeline, wranglerDeployRunner };
|