@tangle-network/agent-eval 0.115.2 → 0.116.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +34 -0
- package/dist/analyst/index.d.ts +8 -10
- package/dist/analyst/index.js +28 -23
- package/dist/analyst/index.js.map +1 -1
- package/dist/{analyst-C8HHvfJp.d.ts → analyst-CFBc14Wc.d.ts} +1 -1
- package/dist/{analyze-runs-BYHg6Irm.d.ts → analyze-runs-0rz_m29H.d.ts} +3 -3
- package/dist/belief-state/index.d.ts +3 -3
- package/dist/benchmarks/index.d.ts +7 -3
- package/dist/benchmarks/index.js +5 -5
- package/dist/campaign/index.d.ts +212 -23
- package/dist/campaign/index.js +20 -5
- package/dist/{chunk-N6MTC3GK.js → chunk-3274WNK7.js} +428 -94
- package/dist/chunk-3274WNK7.js.map +1 -0
- package/dist/{chunk-DRPIZQIT.js → chunk-4D5RVB3W.js} +2 -2
- package/dist/{chunk-LVTGFSHF.js → chunk-7GKEAIAD.js} +2 -2
- package/dist/{chunk-DWLIGZBX.js → chunk-CIUOICJT.js} +748 -3
- package/dist/chunk-CIUOICJT.js.map +1 -0
- package/dist/{chunk-5NVBGKPH.js → chunk-GSW3OBHK.js} +1284 -182
- package/dist/chunk-GSW3OBHK.js.map +1 -0
- package/dist/{chunk-FUCQVFMU.js → chunk-GY4SYVPJ.js} +12 -3
- package/dist/chunk-GY4SYVPJ.js.map +1 -0
- package/dist/chunk-MPHTT5HE.js +74 -0
- package/dist/chunk-MPHTT5HE.js.map +1 -0
- package/dist/{chunk-I2HNIE6N.js → chunk-NBSS5NDZ.js} +4 -4
- package/dist/{chunk-QG5F6463.js → chunk-ONM6PEAE.js} +2 -2
- package/dist/cli.js +2 -2
- package/dist/{code-agent-session-D-g04tcy.d.ts → code-agent-session-CdxteG0y.d.ts} +1 -1
- package/dist/contract/index.d.ts +19 -19
- package/dist/contract/index.js +6 -4
- package/dist/contract/index.js.map +1 -1
- package/dist/{control-CcBiAEnn.d.ts → control-DbcDxouY.d.ts} +1 -1
- package/dist/control.d.ts +2 -2
- package/dist/{default-registry-DltpYR5u.d.ts → default-registry-DDfv22MQ.d.ts} +2 -1
- package/dist/{gepa-dne9JDPL.d.ts → gepa-CQelRtuC.d.ts} +10 -8
- package/dist/hosted/index.d.ts +8 -4
- package/dist/{index-BTEpx9He.d.ts → index-DbCXJfZ1.d.ts} +2 -2
- package/dist/index.d.ts +27 -30
- package/dist/index.js +28 -22
- package/dist/index.js.map +1 -1
- package/dist/{insight-report-IwwvqZZv.d.ts → insight-report-oMVxDTxl.d.ts} +1 -1
- package/dist/{integrity-qemeBAyx.d.ts → integrity-C6PZ73iC.d.ts} +1 -1
- package/dist/kind-factory-DWOvXjR_.d.ts +171 -0
- package/dist/meta-eval/index.d.ts +2 -2
- package/dist/multishot/index.d.ts +6 -2
- package/dist/openapi.json +1 -1
- package/dist/policy-edit-Clb2v6Oa.d.ts +708 -0
- package/dist/{pre-registration-D8h7ZxNL.d.ts → pre-registration--vU0mMtD.d.ts} +4 -4
- package/dist/{provenance-Bibyg1U9.d.ts → provenance-BbVagC68.d.ts} +26 -14
- package/dist/{release-report-CCtzajxP.d.ts → release-report-CamNDe90.d.ts} +2 -2
- package/dist/reporting.d.ts +4 -4
- package/dist/{researcher-Dq-EtpbE.d.ts → researcher-Dwbo_Fxx.d.ts} +5 -5
- package/dist/rl.d.ts +11 -9
- package/dist/rl.js +2 -2
- package/dist/{rubric-predictive-validity-DYTLjGWu.d.ts → rubric-predictive-validity-BIdf9h4R.d.ts} +1 -1
- package/dist/{run-record-B7RTi_ix.d.ts → run-record-CZmcpWPo.d.ts} +1 -1
- package/dist/{runtime-trajectory-Dws7Kpgi.d.ts → runtime-trajectory-CC0jx9ql.d.ts} +1 -1
- package/dist/{semantic-concept-judge-DxJmRkyJ.d.ts → semantic-concept-judge-CKjePUMh.d.ts} +3 -3
- package/dist/{store-C1YxJDEK.d.ts → store-9cAScOcb.d.ts} +132 -1
- package/dist/{summary-report-BJ5aNwZ1.d.ts → summary-report-DTNgQycC.d.ts} +1 -1
- package/dist/traces.d.ts +6 -8
- package/dist/{types-C5gJrOVT.d.ts → types-Ca_63YSD.d.ts} +59 -2
- package/dist/wire/index.js +2 -2
- package/docs/design/loop-taxonomy.md +1 -2
- package/package.json +1 -1
- package/dist/chunk-5NVBGKPH.js.map +0 -1
- package/dist/chunk-AN5UYSVD.js +0 -761
- package/dist/chunk-AN5UYSVD.js.map +0 -1
- package/dist/chunk-DWLIGZBX.js.map +0 -1
- package/dist/chunk-FUCQVFMU.js.map +0 -1
- package/dist/chunk-N6MTC3GK.js.map +0 -1
- package/dist/kind-factory-DcNg13sZ.d.ts +0 -508
- package/dist/llm-client-DyqEH4jH.d.ts +0 -265
- package/dist/policy-edit-RLn8GWof.d.ts +0 -103
- package/dist/raw-provider-sink-C46HDghv.d.ts +0 -132
- /package/dist/{chunk-DRPIZQIT.js.map → chunk-4D5RVB3W.js.map} +0 -0
- /package/dist/{chunk-LVTGFSHF.js.map → chunk-7GKEAIAD.js.map} +0 -0
- /package/dist/{chunk-I2HNIE6N.js.map → chunk-NBSS5NDZ.js.map} +0 -0
- /package/dist/{chunk-QG5F6463.js.map → chunk-ONM6PEAE.js.map} +0 -0
|
@@ -1,8 +1,8 @@
|
|
|
1
1
|
import { D as DatasetScenario, c as Dataset } from './dataset-NENEzRgk.js';
|
|
2
2
|
import { T as TraceStore } from './store-BsVi7ncX.js';
|
|
3
|
-
import { S as Scenario, C as CampaignResult, G as GateResult, c as DispatchFn, b as JudgeConfig, L as LabeledScenarioStore, d as CampaignTraceWriter, e as GenerationRecord, M as MutableSurface, P as ParetoParent, f as SurfaceProposer, g as Gate } from './types-
|
|
3
|
+
import { S as Scenario, C as CampaignResult, G as GateResult, c as DispatchFn, b as JudgeConfig, L as LabeledScenarioStore, d as CampaignTraceWriter, e as GenerationRecord, M as MutableSurface, P as ParetoParent, f as SurfaceProposer, g as Gate } from './types-Ca_63YSD.js';
|
|
4
4
|
import { C as CampaignStorage } from './storage-Dw_f7WMt.js';
|
|
5
|
-
import { L as LlmClientOptions } from './
|
|
5
|
+
import { L as LlmClientOptions } from './policy-edit-Clb2v6Oa.js';
|
|
6
6
|
|
|
7
7
|
/**
|
|
8
8
|
* Pareto frontier — multi-objective optimization over candidate runs.
|
|
@@ -332,10 +332,11 @@ declare function planCampaignRun<TScenario extends Scenario, TArtifact>(opts: Pl
|
|
|
332
332
|
/**
|
|
333
333
|
* `runOptimization` — the improvement loop body. Runs N generations: the
|
|
334
334
|
* `SurfaceProposer` proposes K candidate surfaces per generation, each
|
|
335
|
-
* candidate runs a campaign (the measurement),
|
|
336
|
-
*
|
|
337
|
-
*
|
|
338
|
-
*
|
|
335
|
+
* candidate runs a campaign (the measurement), and only a candidate that beats
|
|
336
|
+
* the single global incumbent becomes the next generation's parent.
|
|
337
|
+
* Proposer-agnostic — the same loop runs an evolutionary population mutator
|
|
338
|
+
* (`evolutionaryProposer`) or any reflective / agentic proposer; they differ
|
|
339
|
+
* only in how `propose()` picks candidates.
|
|
339
340
|
*
|
|
340
341
|
* This is `runLoop`'s shape (plan → measure → decide) specialized to surface
|
|
341
342
|
* improvement: `proposer.propose` = plan, `runCampaign` = the measurement
|
|
@@ -357,7 +358,8 @@ interface RunOptimizationBaseOptions<TScenario extends Scenario, TArtifact> exte
|
|
|
357
358
|
proposer: SurfaceProposer;
|
|
358
359
|
populationSize: number;
|
|
359
360
|
maxGenerations: number;
|
|
360
|
-
/**
|
|
361
|
+
/** @deprecated The loop has one global incumbent and can promote only the
|
|
362
|
+
* single candidate that beats it. Retained for source compatibility. */
|
|
361
363
|
promoteTopK?: number;
|
|
362
364
|
/** DEPTH knob forwarded to the proposer's `propose()` — max iterations the
|
|
363
365
|
* agentic generator may take per candidate. */
|
|
@@ -420,7 +422,7 @@ interface RunOptimizationResult<TArtifact, TScenario extends Scenario> {
|
|
|
420
422
|
paretoFrontier: ParetoParent[];
|
|
421
423
|
}
|
|
422
424
|
/**
|
|
423
|
-
* Improvement loop body: N generations of propose → campaign → rank, maintaining a Pareto frontier and
|
|
425
|
+
* Improvement loop body: N generations of propose → campaign → rank, maintaining a Pareto frontier and one global incumbent across generations.
|
|
424
426
|
*/
|
|
425
427
|
declare function runOptimization<TScenario extends Scenario, TArtifact>(opts: RunOptimizationOptions<TScenario, TArtifact>): Promise<RunOptimizationResult<TArtifact, TScenario>>;
|
|
426
428
|
|
package/dist/hosted/index.d.ts
CHANGED
|
@@ -1,10 +1,14 @@
|
|
|
1
|
-
import { M as MutableSurface, h as GateDecision } from '../types-
|
|
2
|
-
import { I as InsightReport } from '../insight-report-
|
|
3
|
-
import '../
|
|
1
|
+
import { M as MutableSurface, h as GateDecision } from '../types-Ca_63YSD.js';
|
|
2
|
+
import { I as InsightReport } from '../insight-report-oMVxDTxl.js';
|
|
3
|
+
import '../policy-edit-Clb2v6Oa.js';
|
|
4
|
+
import '../run-record-CZmcpWPo.js';
|
|
4
5
|
import '@tangle-network/agent-interface';
|
|
5
6
|
import '../errors-oeQrLqXC.js';
|
|
6
7
|
import '../schema-SGWcK9wa.js';
|
|
7
|
-
import '../
|
|
8
|
+
import '../store-9cAScOcb.js';
|
|
9
|
+
import '../types-C7DGg5ex.js';
|
|
10
|
+
import '@tangle-network/tcloud';
|
|
11
|
+
import '../summary-report-DTNgQycC.js';
|
|
8
12
|
import '../failure-cluster-C48PiReX.js';
|
|
9
13
|
import '../store-BsVi7ncX.js';
|
|
10
14
|
import '../judge-calibration-7C-IDmKr.js';
|
|
@@ -1,5 +1,5 @@
|
|
|
1
|
-
import { S as Scenario, D as DispatchContext, C as CampaignResult } from './types-
|
|
2
|
-
import {
|
|
1
|
+
import { S as Scenario, D as DispatchContext, C as CampaignResult } from './types-Ca_63YSD.js';
|
|
2
|
+
import { a as RunSplitTag } from './run-record-CZmcpWPo.js';
|
|
3
3
|
import { C as CampaignStorage } from './storage-Dw_f7WMt.js';
|
|
4
4
|
|
|
5
5
|
/**
|
package/dist/index.d.ts
CHANGED
|
@@ -1,12 +1,12 @@
|
|
|
1
|
-
export { A as ActionExecutionPolicy, a as ActionPolicyDecision, C as ControlRunToRunRecordOptions, L as LlmJsonCall, b as LlmReviewerConfig, P as ProposeFn, c as ProposeInput, d as ProposeOutput, e as ProposeReviewConfig, f as ProposeReviewControlAction, g as ProposeReviewControlConfig, h as ProposeReviewControlResult, i as ProposeReviewControlState, j as ProposeReviewReport, k as ProposeReviewShot, R as Review, l as ReviewFn, m as ReviewInput, n as ReviewMemoryEntry, o as ReviewMemoryStore, p as RunEvidenceMetadata, V as Verification, q as VerifyFn, r as controlFailureClassFromVerification, s as controlRunToRunRecord, t as createLlmReviewer, u as evaluateActionPolicy, v as inMemoryReviewStore, w as jsonlReviewStore, x as runProposeReview, y as runProposeReviewAsControlLoop, z as scoreFromEvals } from './control-
|
|
2
|
-
import { R as RunRecord,
|
|
3
|
-
export { g as AGENT_PROFILE_KINDS, h as AgentInterfaceProfileLike, A as AgentProfileCell,
|
|
4
|
-
import { B as BehavioralMetrics, A as RunScore, a as RunTrace, E as RunScoreWeights } from './semantic-concept-judge-
|
|
5
|
-
export { G as ConceptComplexity, H as ConceptFinding, J as ConceptSpec, L as ConceptWeightStrategy, C as CreateAnalystAiConfig, M as DEFAULT_COMPLEXITY_WEIGHTS, N as DEFAULT_RUN_SCORE_WEIGHTS, D as DEFAULT_TRACE_ANALYST_KINDS, c as DiffPolicy, F as FAILURE_MODE_KIND_SPEC, g as FindingSubject, h as FindingSubjectKind, j as FindingsDiff, k as FindingsStore, I as IMPROVEMENT_KIND_SPEC, l as KNOWLEDGE_GAP_KIND_SPEC, m as KNOWLEDGE_POISONING_KIND_SPEC, P as PersistedFinding, R as RunCritic, O as RunCriticOptions, Q as SEMANTIC_CONCEPT_JUDGE_VERSION, n as SKILL_USAGE_ANALYST, b as SemanticConceptJudgeInput, S as SemanticConceptJudgeOptions, T as SemanticConceptJudgeResult, o as SkillUsageAnalyst, U as SuboptimalCode, V as SuboptimalSignal, W as aggregateRunScore, X as clamp01, Y as computeTraceMetrics, t as createAnalystAi, Z as createSemanticConceptJudge, u as defaultIsMaterial, v as diffFindings, _ as runSemanticConceptJudge } from './semantic-concept-judge-
|
|
6
|
-
import { m as ChatRequest, q as CreateChatClientOpts } from './
|
|
7
|
-
export {
|
|
8
|
-
export { a as AnalystHooks, A as AnalystRegistry, b as AnalystRegistryOptions, B as BudgetPolicy, D as DefaultAnalystRegistryOptions, R as RegistryRunOpts, c as buildDefaultAnalystRegistry } from './default-registry-
|
|
9
|
-
export {
|
|
1
|
+
export { A as ActionExecutionPolicy, a as ActionPolicyDecision, C as ControlRunToRunRecordOptions, L as LlmJsonCall, b as LlmReviewerConfig, P as ProposeFn, c as ProposeInput, d as ProposeOutput, e as ProposeReviewConfig, f as ProposeReviewControlAction, g as ProposeReviewControlConfig, h as ProposeReviewControlResult, i as ProposeReviewControlState, j as ProposeReviewReport, k as ProposeReviewShot, R as Review, l as ReviewFn, m as ReviewInput, n as ReviewMemoryEntry, o as ReviewMemoryStore, p as RunEvidenceMetadata, V as Verification, q as VerifyFn, r as controlFailureClassFromVerification, s as controlRunToRunRecord, t as createLlmReviewer, u as evaluateActionPolicy, v as inMemoryReviewStore, w as jsonlReviewStore, x as runProposeReview, y as runProposeReviewAsControlLoop, z as scoreFromEvals } from './control-DbcDxouY.js';
|
|
2
|
+
import { R as RunRecord, a as RunSplitTag } from './run-record-CZmcpWPo.js';
|
|
3
|
+
export { g as AGENT_PROFILE_KINDS, h as AgentInterfaceProfileLike, A as AgentProfileCell, e as AgentProfileCellInput, i as AgentProfileCellSchemaVersion, j as AgentProfileCellValidationError, k as AgentProfileDimensionValue, l as AgentProfileHarness, f as AgentProfileJson, m as AgentProfileKind, n as AgentProfileSource, o as AgentProfileSourceInput, J as JudgeScoresRecord, b as RunCostProvenance, d as RunJudgeMetadata, p as RunOutcome, q as RunRecordValidationError, c as RunTokenUsage, r as agentProfileCellHashMaterial, s as agentProfileCellKey, t as assertRunAgentProfileCell, u as buildAgentInterfaceProfileCell, v as buildAgentProfileCell, w as groupRunsByAgentProfileCell, x as isRunRecord, y as modelHasSnapshot, z as parseRunRecordSafe, B as requireAgentProfileCell, C as resolveRunCostProvenance, D as roundTripRunRecord, E as toAgentProfileJson, F as validateAgentProfileCell, G as validateRunRecord, H as verifyAgentProfileCell } from './run-record-CZmcpWPo.js';
|
|
4
|
+
import { B as BehavioralMetrics, A as RunScore, a as RunTrace, E as RunScoreWeights } from './semantic-concept-judge-CKjePUMh.js';
|
|
5
|
+
export { G as ConceptComplexity, H as ConceptFinding, J as ConceptSpec, L as ConceptWeightStrategy, C as CreateAnalystAiConfig, M as DEFAULT_COMPLEXITY_WEIGHTS, N as DEFAULT_RUN_SCORE_WEIGHTS, D as DEFAULT_TRACE_ANALYST_KINDS, c as DiffPolicy, F as FAILURE_MODE_KIND_SPEC, g as FindingSubject, h as FindingSubjectKind, j as FindingsDiff, k as FindingsStore, I as IMPROVEMENT_KIND_SPEC, l as KNOWLEDGE_GAP_KIND_SPEC, m as KNOWLEDGE_POISONING_KIND_SPEC, P as PersistedFinding, R as RunCritic, O as RunCriticOptions, Q as SEMANTIC_CONCEPT_JUDGE_VERSION, n as SKILL_USAGE_ANALYST, b as SemanticConceptJudgeInput, S as SemanticConceptJudgeOptions, T as SemanticConceptJudgeResult, o as SkillUsageAnalyst, U as SuboptimalCode, V as SuboptimalSignal, W as aggregateRunScore, X as clamp01, Y as computeTraceMetrics, t as createAnalystAi, Z as createSemanticConceptJudge, u as defaultIsMaterial, v as diffFindings, _ as runSemanticConceptJudge } from './semantic-concept-judge-CKjePUMh.js';
|
|
6
|
+
import { L as LlmClientOptions, m as ChatRequest, q as CreateChatClientOpts } from './policy-edit-Clb2v6Oa.js';
|
|
7
|
+
export { A as Analyst, b as AnalystContext, h as AnalystCost, d as AnalystFinding, j as AnalystInputKind, k as AnalystRequirements, g as AnalystRunEvent, f as AnalystRunInputs, e as AnalystRunResult, c as AnalystRunSummary, i as AnalystSeverity, l as ChatCallOpts, C as ChatClient, n as ChatResponse, o as ChatTransport, p as CliBridgeTransportOpts, D as DirectProviderTransportOpts, E as EvidenceRef, F as FindingToPolicyEditOptions, a5 as LlmCallError, a6 as LlmCallRequest, a7 as LlmCallResult, a8 as LlmClient, a9 as LlmMessage, aa as LlmRouteAssertionError, a as LlmRouteRequirements, ab as LlmUsage, M as MockTransportOpts, r as POLICY_EDIT_AXES, s as POLICY_EDIT_CANDIDATE_RECORD_SCHEMA, t as POLICY_EDIT_TARGET_SURFACES, u as PolicyEdit, v as PolicyEditAdmission, w as PolicyEditAdmissionOptions, x as PolicyEditAxis, P as PolicyEditCandidateRecord, y as PolicyEditChange, z as PolicyEditExpectedGain, B as PolicyEditGainDirection, G as PolicyEditGainUnit, H as PolicyEditInit, I as PolicyEditRisk, J as PolicyEditSchemaVersion, K as PolicyEditSource, N as PolicyEditTarget, O as PolicyEditTargetSurface, Q as PolicyEditValidationError, R as RouterTransportOpts, S as SandboxSdkTransportOpts, T as admitPolicyEdit, U as applyPolicyEditToSurface, ac as assertLlmRoute, ad as backoffMs, ae as callLlm, af as callLlmJson, V as computeFindingId, W as computePolicyEditId, X as createChatClient, Y as isPolicyEdit, ag as isTransientLlmError, Z as makeFinding, _ as makePolicyEdit, $ as makePolicyEditCandidateRecord, a0 as policyEditFromFinding, a1 as policyEditsFromFindings, ah as probeLlm, a2 as scorePolicyEditReadiness, ai as stripFencedJson, a3 as validatePolicyEdit, a4 as validatePolicyEditCandidateRecord } from './policy-edit-Clb2v6Oa.js';
|
|
8
|
+
export { a as AnalystHooks, A as AnalystRegistry, b as AnalystRegistryOptions, B as BudgetPolicy, D as DefaultAnalystRegistryOptions, R as RegistryRunOpts, c as buildDefaultAnalystRegistry } from './default-registry-DDfv22MQ.js';
|
|
9
|
+
export { C as CreateTraceAnalystKindOpts, a as RawAnalystFinding, c as TraceAnalystGolden, T as TraceAnalystKindSpec, d as createTraceAnalystKind, r as renderPriorFindings } from './kind-factory-DWOvXjR_.js';
|
|
10
10
|
import { TCloud } from '@tangle-network/tcloud';
|
|
11
11
|
import { B as BenchmarkRunnerConfig, S as Scenario, c as BenchmarkReport, P as ProductClientConfig, C as CheckResult, T as TestResult, d as PersonaConfig, D as DriverResult, e as DriverState, b as JudgeFn, f as CollectedArtifacts, g as ScenarioResult, h as TurnMetrics, i as ScenarioFile, j as CompletionCriterion } from './types-C7DGg5ex.js';
|
|
12
12
|
export { A as ArtifactCheck, k as ArtifactResult, E as EvalResult, F as FeedbackPattern, l as JudgeConfig, a as JudgeInput, m as JudgeRubric, J as JudgeScore, n as PersonaRigor, R as RouteMap, o as RubricDimension, p as Turn, q as TurnResult } from './types-C7DGg5ex.js';
|
|
@@ -16,32 +16,31 @@ import { F as FailureClass, T as ToolSpan, h as BudgetSpec, B as BudgetLedgerEnt
|
|
|
16
16
|
export { A as Artifact, E as EventKind, i as FAILURE_CLASSES, G as GenericSpan, J as JudgeSpan, M as Message, c as RetrievalSpan, g as RunLayer, f as RunStatus, d as SandboxSpan, j as SpanBase, b as SpanKind, k as SpanStatus, l as TRACE_SCHEMA_VERSION, e as TraceEvent, m as isJudgeSpan, n as isLlmSpan, o as isRetrievalSpan, p as isSandboxSpan, q as isToolSpan } from './schema-SGWcK9wa.js';
|
|
17
17
|
import { A as AgentEvalError, J as JudgeError, a as ConfigError } from './errors-oeQrLqXC.js';
|
|
18
18
|
export { b as AgentEvalErrorCode, C as CaptureIntegrityError, N as NotFoundError, R as ReplayError, V as ValidationError, c as VerificationError } from './errors-oeQrLqXC.js';
|
|
19
|
-
import { c as CorrectnessChecker } from './pre-registration
|
|
20
|
-
export { A as ArtifactCheckArtifact, e as ArtifactEventLike, f as ArtifactValidator, g as BackendIntegrityError, B as BackendIntegrityReport, h as ComparePairedArmsOptions, C as CompletionRequirement, a as CompletionVerdict, H as HypothesisManifest, i as HypothesisResult, j as LlmCorrectnessCheckerOpts, L as LlmJudgeDimension, d as LlmJudgeOptions, M as MatchedPair, k as PairArmsOptions, m as PairArmsResult, n as PairedArmRow, P as PairedArmsComparison, o as PairedCorrectness, p as PairedMetricDelta, q as ProducedProposal, b as ProducedState, r as ProposalEventLike, s as RequirementCheck, R as RuntimeEventLike, t as SatisfiedBy, S as SignedManifest, u as SignedManifestAlgo, T as TaskGold, v as ToolCallEventLike, V as ValidationContext, w as ValidationIssue, x as ValidationResult, y as assertRealBackend, z as byteLengthRange, D as canonicalize, E as comparePairedArms, F as completionVerdict, G as composeValidators, I as containsAll, J as createLlmCorrectnessChecker, K as createTokenRecallChecker, N as evaluateHypothesis, O as extractProducedState, Q as hashJson, U as jsonHasKeys, l as llmJudge, W as pairArms, X as parseCorrectnessResponse, Y as regexMatch, Z as signManifest, _ as summarizeBackendIntegrity, $ as verifyCompletion, a0 as verifyManifest } from './pre-registration
|
|
19
|
+
import { c as CorrectnessChecker } from './pre-registration--vU0mMtD.js';
|
|
20
|
+
export { A as ArtifactCheckArtifact, e as ArtifactEventLike, f as ArtifactValidator, g as BackendIntegrityError, B as BackendIntegrityReport, h as ComparePairedArmsOptions, C as CompletionRequirement, a as CompletionVerdict, H as HypothesisManifest, i as HypothesisResult, j as LlmCorrectnessCheckerOpts, L as LlmJudgeDimension, d as LlmJudgeOptions, M as MatchedPair, k as PairArmsOptions, m as PairArmsResult, n as PairedArmRow, P as PairedArmsComparison, o as PairedCorrectness, p as PairedMetricDelta, q as ProducedProposal, b as ProducedState, r as ProposalEventLike, s as RequirementCheck, R as RuntimeEventLike, t as SatisfiedBy, S as SignedManifest, u as SignedManifestAlgo, T as TaskGold, v as ToolCallEventLike, V as ValidationContext, w as ValidationIssue, x as ValidationResult, y as assertRealBackend, z as byteLengthRange, D as canonicalize, E as comparePairedArms, F as completionVerdict, G as composeValidators, I as containsAll, J as createLlmCorrectnessChecker, K as createTokenRecallChecker, N as evaluateHypothesis, O as extractProducedState, Q as hashJson, U as jsonHasKeys, l as llmJudge, W as pairArms, X as parseCorrectnessResponse, Y as regexMatch, Z as signManifest, _ as summarizeBackendIntegrity, $ as verifyCompletion, a0 as verifyManifest } from './pre-registration--vU0mMtD.js';
|
|
21
21
|
import { T as TraceEmitter } from './emitter-BRchAAAx.js';
|
|
22
22
|
export { R as RunCompleteHook, a as RunCompleteHookContext, S as SpanHandle, b as TraceEmitterOptions, l as llmSpanFromProvider } from './emitter-BRchAAAx.js';
|
|
23
|
-
import { h as ReleaseConfidenceThresholds, f as ReleaseConfidenceScorecard } from './release-report-
|
|
24
|
-
export { A as ActionableSideInfo, o as AsiSeverity, B as BootstrapOptions, a as BootstrapResult, J as JudgeReplayGateArgs, R as ReleaseConfidenceAxis, b as ReleaseConfidenceAxisName, c as ReleaseConfidenceInput, d as ReleaseConfidenceIssue, e as ReleaseConfidenceMetrics, g as ReleaseConfidenceStatus, i as ReleaseTraceEvidence, j as RenderReleaseReportOptions, V as Verdict, k as assertReleaseConfidence, l as bootstrapCi, m as evaluateReleaseConfidence, n as judgeReplayGate, r as renderReleaseReport } from './release-report-
|
|
23
|
+
import { h as ReleaseConfidenceThresholds, f as ReleaseConfidenceScorecard } from './release-report-CamNDe90.js';
|
|
24
|
+
export { A as ActionableSideInfo, o as AsiSeverity, B as BootstrapOptions, a as BootstrapResult, J as JudgeReplayGateArgs, R as ReleaseConfidenceAxis, b as ReleaseConfidenceAxisName, c as ReleaseConfidenceInput, d as ReleaseConfidenceIssue, e as ReleaseConfidenceMetrics, g as ReleaseConfidenceStatus, i as ReleaseTraceEvidence, j as RenderReleaseReportOptions, V as Verdict, k as assertReleaseConfidence, l as bootstrapCi, m as evaluateReleaseConfidence, n as judgeReplayGate, r as renderReleaseReport } from './release-report-CamNDe90.js';
|
|
25
25
|
export { c as CliffsMagnitude, d as CorpusAgreementOptions, e as CorpusAgreementPerDimension, C as CorpusAgreementReport, f as CorpusScoreRecord, g as EProcess, h as EProcessOptions, E as EProcessState, i as EProcessStep, M as McNemarResult, P as PairedBootstrapOptions, a as PairedBootstrapResult, j as PairedSignTestResult, k as ProportionInterval, R as RiskDifferenceResult, S as SignTestAlternative, W as WeightedCompositeInput, l as WeightedCompositeResult, b as benjaminiHochberg, m as bonferroni, n as cliffsDelta, o as cohensD, q as confidenceInterval, r as corpusInterRaterAgreement, s as corpusInterRaterAgreementFromJudgeScores, t as eProcess, u as holm, v as interRaterReliability, x as interpretCliffs, y as mannWhitneyU, z as mcnemar, A as mcnemarPower, B as mcnemarRequiredN, D as mulberry32, F as normalizeScores, p as pairedBootstrap, G as pairedMde, H as pairedRiskDifference, I as pairedSignTest, J as pairedTTest, K as partialCredit, L as passAtK, N as pearsonR, O as ranks, Q as requiredSampleSize, T as spearmanR, U as weightedComposite, V as weightedMean, w as wilcoxonSignedRank, X as wilson } from './statistics-oUbOJe-S.js';
|
|
26
26
|
import { OtelExporter, OtelExportConfig } from './traces.js';
|
|
27
27
|
export { CaptureFetchContext, CaptureFetchOptions, DEFAULT_REDACTION_RULES, ExportableSpan, ExtractedUsage, FlattenOtlpOptions, LLM_CACHED_TOKENS, LLM_CACHED_TOKEN_ATTR_KEYS, LLM_COST_ATTR_KEYS, LLM_COST_USD, LLM_INPUT_TOKENS, LLM_INPUT_TOKEN_ATTR_KEYS, LLM_MODEL_ATTR_KEYS, LLM_MODEL_NAME, LLM_OUTPUT_TOKENS, LLM_OUTPUT_TOKEN_ATTR_KEYS, OPENINFERENCE_SPAN_KIND, OTEL_AGENT_EVAL_SCOPE, OtlpExport, OtlpFileTraceStore, OtlpFileTraceStoreOptions, OtlpFlatLine, OtlpResourceSpans, OtlpSpan, OtlpToRunRecordsOptions, OtlpTraceRunRecord, ProjectedOtlpSpan, REDACTION_VERSION, RedactionReport, RedactionRule, ReplayCache, ReplayCacheEntry, ReplayCacheMissError, ReplayCacheStats, ReplayFetchOptions, SPAN_KIND_ATTR_KEYS, SpanNotFoundError, TOOL_NAME, TOOL_NAME_ATTR_KEYS, TRACE_ANALYST_ACTOR_DESCRIPTION, TRACE_ANALYST_ACTOR_DESCRIPTION_VERSION, TRACE_ANALYST_SUBAGENT_DESCRIPTION, TraceAggregate, TraceAnalystHookOptions, TraceFileMissingError, TraceInsightContext, TraceInsightFinding, TraceInsightPanelRole, TraceInsightPromptInput, TraceInsightQualityGate, TraceInsightQuestion, TraceInsightReadiness, TraceInsightSuite, TraceInsightTask, TraceNotFoundError, TraceStoreSource, TraceStoreToOtlpOptions, TracesToOtlpResult, asNumber, asString, buildTraceAnalystTools, buildTraceInsightContext, buildTraceInsightPrompt, captureFetchToRawSink, convertTraceStoresToOtlp, createOtelExporter, createOtelTracingStore, createReplayFetch, defaultTraceInsightPanel, describeTraceInsightScope, domainEvidencePattern, exportRunAsOtlp, extractOtlpAttributes, extractUsage, extractUsageFromResponse, extractUsageFromSse, firstNumberAttr, firstStringAttr, flattenOtlpExportToNdjson, inferDomainKeywords, inferOtlpKind, iterateRawCalls, otelRunCompleteHook, otlpToRunRecords, otlpToTraceRunRecords, planTraceInsightQuestions, projectOtlpFlatLine, readOtlpStatus, redactString, redactValue, scoreTraceInsightReadiness, stringField, tokenizeDomainWords, traceAnalystFunctionGroup, traceAnalystOnRunComplete, traceSpanKindToOpenInferenceKind } from './traces.js';
|
|
28
|
-
import { a as AnalyzeTracesInput, A as AnalyzeTracesOptions, b as AnalyzeTracesResult } from './analyst-
|
|
29
|
-
export { c as AnalyzeTracesTurnSnapshot, d as analyzeTraces } from './analyst-
|
|
30
|
-
import { a as TraceAnalystSpan } from './store-
|
|
31
|
-
export { D as DEFAULT_TRACE_ANALYST_BUDGETS, b as DatasetOverview, E as ErrorCluster, Q as QueryTracesPage, S as SearchSpanResult,
|
|
32
|
-
import { b as JudgeConfig, J as JudgeScore, S as Scenario$1, g as Gate } from './types-
|
|
33
|
-
import { A as AnalyzeRunsOptions } from './analyze-runs-
|
|
34
|
-
import { q as Objective, s as ParetoResult, h as GepaProposerConstraints, a as RunImprovementLoopResult } from './gepa-
|
|
35
|
-
export { t as DEFAULT_RED_TEAM_CORPUS, D as Direction, e as RedTeamCase, u as RedTeamCategory, v as RedTeamFinding, w as RedTeamPayload, x as RedTeamReport, y as crowdingDistance, z as dominates, A as paretoFrontier, B as paretoFrontierWithCrowding, E as redTeamDataset, F as redTeamReport, H as scalarScore, I as scoreRedTeamOutput, J as toolNamesForRun } from './gepa-
|
|
28
|
+
import { a as AnalyzeTracesInput, A as AnalyzeTracesOptions, b as AnalyzeTracesResult } from './analyst-CFBc14Wc.js';
|
|
29
|
+
export { c as AnalyzeTracesTurnSnapshot, d as analyzeTraces } from './analyst-CFBc14Wc.js';
|
|
30
|
+
import { a as TraceAnalystSpan } from './store-9cAScOcb.js';
|
|
31
|
+
export { D as DEFAULT_TRACE_ANALYST_BUDGETS, b as DatasetOverview, E as ErrorCluster, F as FileSystemRawProviderSink, c as FileSystemRawProviderSinkOptions, I as InMemoryRawProviderSink, d as InMemoryRawProviderSinkOptions, N as NoopRawProviderSink, P as ProviderRedactor, Q as QueryTracesPage, e as RawProviderDirection, f as RawProviderEvent, R as RawProviderSink, g as RawProviderSinkFilter, S as SearchSpanResult, h as SearchTraceResult, i as SpanMatchRecord, j as TRACE_ANALYST_TRUNCATION_MARKER_PREFIX, T as TraceAnalysisStore, k as TraceAnalystByteBudgets, l as TraceAnalystFilters, m as TraceAnalystSpanKind, n as TraceAnalystSpanStatus, o as TraceAnalystTraceSummary, V as ViewSpansResult, p as ViewTraceOversized, q as ViewTraceResult, r as defaultProviderRedactor, s as providerFromBaseUrl } from './store-9cAScOcb.js';
|
|
32
|
+
import { b as JudgeConfig, J as JudgeScore, S as Scenario$1, g as Gate } from './types-Ca_63YSD.js';
|
|
33
|
+
import { A as AnalyzeRunsOptions } from './analyze-runs-0rz_m29H.js';
|
|
34
|
+
import { q as Objective, s as ParetoResult, h as GepaProposerConstraints, a as RunImprovementLoopResult } from './gepa-CQelRtuC.js';
|
|
35
|
+
export { t as DEFAULT_RED_TEAM_CORPUS, D as Direction, e as RedTeamCase, u as RedTeamCategory, v as RedTeamFinding, w as RedTeamPayload, x as RedTeamReport, y as crowdingDistance, z as dominates, A as paretoFrontier, B as paretoFrontierWithCrowding, E as redTeamDataset, F as redTeamReport, H as scalarScore, I as scoreRedTeamOutput, J as toolNamesForRun } from './gepa-CQelRtuC.js';
|
|
36
36
|
import { S as SandboxDriver, H as HarnessConfig, a as SandboxHarnessResult } from './test-graded-scenario-mzYBKspu.js';
|
|
37
37
|
export { D as DockerSandboxDriver, c as SandboxHarness, d as SandboxResult, e as SubprocessSandboxDriver, f as SubprocessSandboxDriverOptions, g as TestGradedRunOptions, b as TestGradedRunResult, T as TestGradedScenario, h as TestOutputParser, i as composeParsers, j as jestTestParser, p as pytestTestParser, r as runTestGradedScenario, v as vitestTestParser } from './test-graded-scenario-mzYBKspu.js';
|
|
38
|
-
export { b as RunIntegrityError, R as RunIntegrityExpectations, c as RunIntegrityIssue, d as RunIntegrityIssueCode, a as RunIntegrityReport, e as assertRunCaptured, t as throwIfRunIncomplete } from './integrity-
|
|
38
|
+
export { b as RunIntegrityError, R as RunIntegrityExpectations, c as RunIntegrityIssue, d as RunIntegrityIssueCode, a as RunIntegrityReport, e as assertRunCaptured, t as throwIfRunIncomplete } from './integrity-C6PZ73iC.js';
|
|
39
39
|
export { a as aggregateLlm, b as argHash, g as groupBy, j as judgeSpans, l as llmSpans, r as runFailureClass, c as runsForScenario, t as toolSpans } from './query-Ck190MOd.js';
|
|
40
|
-
export { F as FileSystemRawProviderSink, a as FileSystemRawProviderSinkOptions, I as InMemoryRawProviderSink, b as InMemoryRawProviderSinkOptions, N as NoopRawProviderSink, P as ProviderRedactor, c as RawProviderDirection, d as RawProviderEvent, R as RawProviderSink, e as RawProviderSinkFilter, f as defaultProviderRedactor, p as providerFromBaseUrl } from './raw-provider-sink-C46HDghv.js';
|
|
41
40
|
import { T as TraceStore, R as RunFilter } from './store-BsVi7ncX.js';
|
|
42
41
|
export { E as EventFilter, F as FileSystemTraceStore, a as FileSystemTraceStoreOptions, I as InMemoryTraceStore, S as SpanFilter } from './store-BsVi7ncX.js';
|
|
43
42
|
export { D as DEFAULT_FAILURE_RULES, b as FailureClassification, c as FailureContext, d as FailureRule, e as classifyFailure } from './failure-cluster-C48PiReX.js';
|
|
44
|
-
export { P as ProjectRuntimeTrajectoryEvidenceOptions, a as RuntimeTrajectoryEvidenceProjection, b as RuntimeTrajectoryEvidenceSummary, c as RuntimeTrajectoryHookEvent, R as RuntimeTrajectoryRecord, d as RuntimeTrajectoryRunRecord, p as parseRuntimeTrajectoryHookEvent, e as projectRuntimeTrajectoryEvidence } from './runtime-trajectory-
|
|
43
|
+
export { P as ProjectRuntimeTrajectoryEvidenceOptions, a as RuntimeTrajectoryEvidenceProjection, b as RuntimeTrajectoryEvidenceSummary, c as RuntimeTrajectoryHookEvent, R as RuntimeTrajectoryRecord, d as RuntimeTrajectoryRunRecord, p as parseRuntimeTrajectoryHookEvent, e as projectRuntimeTrajectoryEvidence } from './runtime-trajectory-CC0jx9ql.js';
|
|
45
44
|
import { a as BaselineReport, b as Trajectory, T as TrajectoryStep } from './baseline-DsNteOgR.js';
|
|
46
45
|
export { B as BaselineOptions, M as MetricSamples, d as MetricVerdict, e as ToolStats, f as ToolUseMetrics, g as ToolUseOptions, h as buildTrajectory, i as compareToBaseline, c as computeToolUseMetrics, j as iqr, w as welchsTTest } from './baseline-DsNteOgR.js';
|
|
47
46
|
import { HarnessType, AgentProfile } from '@tangle-network/agent-interface';
|
|
@@ -55,15 +54,13 @@ export { d as DatasetDifficulty, b as DatasetManifest, e as DatasetProvenance, a
|
|
|
55
54
|
export { a as CalibrationResult, c as CandidateScore, C as ContinuousAgreement, d as ContinuousAgreementOptions, b as ContinuousCalibrationResult, G as GoldenItem, P as PositionalBiasResult, S as SelfPreferenceResult, V as VerbosityBiasResult, e as calibrateJudge, f as calibrateJudgeContinuous, g as continuousAgreement, p as positionalBias, s as selfPreference, v as verbosityBias } from './judge-calibration-7C-IDmKr.js';
|
|
56
55
|
import { L as Layer, S as Severity, b as LayerResult, c as VerifyContext } from './multi-layer-verifier-BsqKuLyN.js';
|
|
57
56
|
export { F as Finding, d as LayerStatus, M as MultiLayerVerifier, a as VerificationReport, V as VerifyOptions, g as gradeSemanticStatus } from './multi-layer-verifier-BsqKuLyN.js';
|
|
58
|
-
|
|
59
|
-
export { d as
|
|
60
|
-
export {
|
|
61
|
-
export { C as CallbackResearcher, d as CallbackResearcherOptions, e as CampaignFactoryParams, f as CampaignIntegrityPolicy, g as CampaignRunContext, h as CampaignRunOutcome, i as CampaignRunner, j as CampaignScenario, k as CampaignVariant, c as EvalCampaignOptions, b as EvalCampaignResult, E as ExperimentPlan, a as ExperimentResult, l as FailedRun, F as FailureMode, N as NoopResearcher, R as Researcher, S as SteeringChange, r as runEvalCampaign } from './researcher-Dq-EtpbE.js';
|
|
62
|
-
export { G as GainDistributionBin, a as GainDistributionFigureSpec, b as GainDistributionOptions, m as GateDecision, n as GateEvidence, H as HeldOutGate, o as HeldOutGateConfig, q as HeldOutGateRejectionCode, P as ParetoFigureSpec, c as ParetoPoint, R as RESEARCH_REPORT_HARD_PAIR_FLOOR, d as ResearchReport, e as ResearchReportCandidate, f as ResearchReportDecision, g as ResearchReportMethodology, h as ResearchReportOptions, i as ResearchReportRecommendation, S as SummaryTable, j as SummaryTableOptions, k as SummaryTableRow, l as gainHistogram, p as paretoChart, r as researchReport, s as summaryTable } from './summary-report-BJ5aNwZ1.js';
|
|
57
|
+
export { B as BENCHMARK_SPLIT_SEED, a as BenchmarkAdapter, b as BenchmarkDatasetItem, c as BenchmarkEvaluation, d as BenchmarkFamily, e as BenchmarkResponder, f as BenchmarkScenario, g as BenchmarkSource, h as BenchmarkTaskKind, i as benchmarkDeterministicSplit, j as benchmarks } from './index-DbCXJfZ1.js';
|
|
58
|
+
export { C as CallbackResearcher, d as CallbackResearcherOptions, e as CampaignFactoryParams, f as CampaignIntegrityPolicy, g as CampaignRunContext, h as CampaignRunOutcome, i as CampaignRunner, j as CampaignScenario, k as CampaignVariant, c as EvalCampaignOptions, b as EvalCampaignResult, E as ExperimentPlan, a as ExperimentResult, l as FailedRun, F as FailureMode, N as NoopResearcher, R as Researcher, S as SteeringChange, r as runEvalCampaign } from './researcher-Dwbo_Fxx.js';
|
|
59
|
+
export { G as GainDistributionBin, a as GainDistributionFigureSpec, b as GainDistributionOptions, m as GateDecision, n as GateEvidence, H as HeldOutGate, o as HeldOutGateConfig, q as HeldOutGateRejectionCode, P as ParetoFigureSpec, c as ParetoPoint, R as RESEARCH_REPORT_HARD_PAIR_FLOOR, d as ResearchReport, e as ResearchReportCandidate, f as ResearchReportDecision, g as ResearchReportMethodology, h as ResearchReportOptions, i as ResearchReportRecommendation, S as SummaryTable, j as SummaryTableOptions, k as SummaryTableRow, l as gainHistogram, p as paretoChart, r as researchReport, s as summaryTable } from './summary-report-DTNgQycC.js';
|
|
63
60
|
export { I as InterimReleaseConfidence, a as InterimReleaseConfidenceInput, P as PairedEvalueOptions, b as PairedEvalueSequence, c as PairedEvalueStep, S as SequentialDecision, e as evaluateInterimReleaseConfidence, p as pairedEvalueSequence } from './sequential-5iSVfzl2.js';
|
|
64
61
|
import '@ax-llm/ax';
|
|
65
62
|
import 'zod';
|
|
66
|
-
import './insight-report-
|
|
63
|
+
import './insight-report-oMVxDTxl.js';
|
|
67
64
|
import './storage-Dw_f7WMt.js';
|
|
68
65
|
|
|
69
66
|
/**
|
package/dist/index.js
CHANGED
|
@@ -61,7 +61,7 @@ import {
|
|
|
61
61
|
pairArms,
|
|
62
62
|
parseCorrectnessResponse,
|
|
63
63
|
verifyCompletion
|
|
64
|
-
} from "./chunk-
|
|
64
|
+
} from "./chunk-GSW3OBHK.js";
|
|
65
65
|
import {
|
|
66
66
|
DEFAULT_MUTATION_PRIMITIVES,
|
|
67
67
|
DEFAULT_RED_TEAM_CORPUS,
|
|
@@ -85,7 +85,7 @@ import {
|
|
|
85
85
|
scoreRedTeamOutput,
|
|
86
86
|
surfaceContentHash,
|
|
87
87
|
toolNamesForRun
|
|
88
|
-
} from "./chunk-
|
|
88
|
+
} from "./chunk-3274WNK7.js";
|
|
89
89
|
import {
|
|
90
90
|
MODEL_PRICING,
|
|
91
91
|
MetricsCollector,
|
|
@@ -119,29 +119,17 @@ import {
|
|
|
119
119
|
defaultIsMaterial,
|
|
120
120
|
diffFindings,
|
|
121
121
|
runSemanticConceptJudge
|
|
122
|
-
} from "./chunk-
|
|
122
|
+
} from "./chunk-NBSS5NDZ.js";
|
|
123
123
|
import {
|
|
124
124
|
buildDefaultAnalystRegistry,
|
|
125
125
|
computeTraceMetrics
|
|
126
|
-
} from "./chunk-
|
|
126
|
+
} from "./chunk-7GKEAIAD.js";
|
|
127
127
|
import {
|
|
128
128
|
DEFAULT_RUN_SCORE_WEIGHTS,
|
|
129
129
|
Mutex,
|
|
130
|
-
POLICY_EDIT_AXES,
|
|
131
|
-
POLICY_EDIT_TARGET_SURFACES,
|
|
132
|
-
PolicyEditValidationError,
|
|
133
|
-
admitPolicyEdit,
|
|
134
130
|
aggregateRunScore,
|
|
135
|
-
|
|
136
|
-
|
|
137
|
-
computePolicyEditId,
|
|
138
|
-
isPolicyEdit,
|
|
139
|
-
makePolicyEdit,
|
|
140
|
-
policyEditFromFinding,
|
|
141
|
-
policyEditsFromFindings,
|
|
142
|
-
scorePolicyEditReadiness,
|
|
143
|
-
validatePolicyEdit
|
|
144
|
-
} from "./chunk-AN5UYSVD.js";
|
|
131
|
+
clamp01
|
|
132
|
+
} from "./chunk-MPHTT5HE.js";
|
|
145
133
|
import {
|
|
146
134
|
AnalystRegistry,
|
|
147
135
|
DEFAULT_TRACE_ANALYST_KINDS,
|
|
@@ -149,11 +137,26 @@ import {
|
|
|
149
137
|
IMPROVEMENT_KIND_SPEC,
|
|
150
138
|
KNOWLEDGE_GAP_KIND_SPEC,
|
|
151
139
|
KNOWLEDGE_POISONING_KIND_SPEC,
|
|
140
|
+
POLICY_EDIT_AXES,
|
|
141
|
+
POLICY_EDIT_CANDIDATE_RECORD_SCHEMA,
|
|
142
|
+
POLICY_EDIT_TARGET_SURFACES,
|
|
143
|
+
PolicyEditValidationError,
|
|
144
|
+
admitPolicyEdit,
|
|
145
|
+
applyPolicyEditToSurface,
|
|
152
146
|
computeFindingId,
|
|
147
|
+
computePolicyEditId,
|
|
153
148
|
createTraceAnalystKind,
|
|
149
|
+
isPolicyEdit,
|
|
154
150
|
makeFinding,
|
|
155
|
-
|
|
156
|
-
|
|
151
|
+
makePolicyEdit,
|
|
152
|
+
makePolicyEditCandidateRecord,
|
|
153
|
+
policyEditFromFinding,
|
|
154
|
+
policyEditsFromFindings,
|
|
155
|
+
renderPriorFindings,
|
|
156
|
+
scorePolicyEditReadiness,
|
|
157
|
+
validatePolicyEdit,
|
|
158
|
+
validatePolicyEditCandidateRecord
|
|
159
|
+
} from "./chunk-CIUOICJT.js";
|
|
157
160
|
import {
|
|
158
161
|
allCriticalPassed,
|
|
159
162
|
controlFailureClassFromVerification,
|
|
@@ -184,7 +187,7 @@ import {
|
|
|
184
187
|
} from "./chunk-MOXWMGPC.js";
|
|
185
188
|
import {
|
|
186
189
|
runEvalCampaign
|
|
187
|
-
} from "./chunk-
|
|
190
|
+
} from "./chunk-ONM6PEAE.js";
|
|
188
191
|
import "./chunk-ARU2PZFM.js";
|
|
189
192
|
import {
|
|
190
193
|
evaluateInterimReleaseConfidence,
|
|
@@ -381,7 +384,7 @@ import {
|
|
|
381
384
|
isTransientLlmError,
|
|
382
385
|
probeLlm,
|
|
383
386
|
stripFencedJson
|
|
384
|
-
} from "./chunk-
|
|
387
|
+
} from "./chunk-GY4SYVPJ.js";
|
|
385
388
|
import {
|
|
386
389
|
FileSystemRawProviderSink,
|
|
387
390
|
InMemoryRawProviderSink,
|
|
@@ -11523,6 +11526,7 @@ export {
|
|
|
11523
11526
|
OTEL_AGENT_EVAL_SCOPE,
|
|
11524
11527
|
OtlpFileTraceStore,
|
|
11525
11528
|
POLICY_EDIT_AXES,
|
|
11529
|
+
POLICY_EDIT_CANDIDATE_RECORD_SCHEMA,
|
|
11526
11530
|
POLICY_EDIT_TARGET_SURFACES,
|
|
11527
11531
|
PairwiseSteeringOptimizer,
|
|
11528
11532
|
PolicyEditValidationError,
|
|
@@ -11832,6 +11836,7 @@ export {
|
|
|
11832
11836
|
makeEvalTools,
|
|
11833
11837
|
makeFinding,
|
|
11834
11838
|
makePolicyEdit,
|
|
11839
|
+
makePolicyEditCandidateRecord,
|
|
11835
11840
|
mannWhitneyU,
|
|
11836
11841
|
matchGoldens,
|
|
11837
11842
|
matchSpan,
|
|
@@ -12009,6 +12014,7 @@ export {
|
|
|
12009
12014
|
userQuestionsForKnowledgeGaps,
|
|
12010
12015
|
validateAgentProfileCell,
|
|
12011
12016
|
validatePolicyEdit,
|
|
12017
|
+
validatePolicyEditCandidateRecord,
|
|
12012
12018
|
validateProductBenchmarkManifest,
|
|
12013
12019
|
validateProductBenchmarkRecord,
|
|
12014
12020
|
validateProductBenchmarkRun,
|