@tangle-network/agent-eval 0.102.0 → 0.103.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/dist/adapters/http.d.ts +2 -2
- package/dist/adapters/langchain.d.ts +2 -2
- package/dist/adapters/otel.d.ts +4 -4
- package/dist/analyst/index.d.ts +8 -8
- package/dist/{analyze-runs-BlJRBniC.d.ts → analyze-runs-Cd-A_K4l.d.ts} +3 -3
- package/dist/belief-state/index.d.ts +3 -3
- package/dist/benchmarks/index.d.ts +2 -2
- package/dist/campaign/index.d.ts +14 -14
- package/dist/campaign/index.js +34 -11
- package/dist/campaign/index.js.map +1 -1
- package/dist/{chunk-G6S73VA7.js → chunk-2NSLDY4B.js} +3 -2
- package/dist/{chunk-G6S73VA7.js.map → chunk-2NSLDY4B.js.map} +1 -1
- package/dist/{chunk-4LWD6GC7.js → chunk-6FIAJHCU.js} +2 -2
- package/dist/{chunk-PMF5WIBX.js → chunk-7RBJANJD.js} +2 -2
- package/dist/{chunk-BOETF6BU.js → chunk-B2TMQM62.js} +2 -2
- package/dist/{chunk-QUCGGMYM.js → chunk-HV5PBTJF.js} +3 -3
- package/dist/{chunk-CMJSTXUR.js → chunk-IXOV77YF.js} +105 -18
- package/dist/chunk-IXOV77YF.js.map +1 -0
- package/dist/{chunk-52CCCXU3.js → chunk-NTVWIH24.js} +104 -43
- package/dist/chunk-NTVWIH24.js.map +1 -0
- package/dist/chunk-RQNOLV3I.js +855 -0
- package/dist/chunk-RQNOLV3I.js.map +1 -0
- package/dist/{chunk-JCUREYF5.js → chunk-U3IDYATS.js} +2 -2
- package/dist/{chunk-LSCBODPQ.js → chunk-XKA6ZGEY.js} +11 -2
- package/dist/chunk-XKA6ZGEY.js.map +1 -0
- package/dist/{code-agent-session-B6ZcDwyA.d.ts → code-agent-session-Ce-9u7YM.d.ts} +1 -1
- package/dist/contract/index.d.ts +16 -16
- package/dist/contract/index.js +5 -5
- package/dist/{control-DC8TELh0.d.ts → control-C8RmK9H4.d.ts} +1 -1
- package/dist/control.d.ts +2 -2
- package/dist/control.js +2 -2
- package/dist/{corpus-ONOzGFmG.d.ts → corpus-CiSzzLa5.d.ts} +1 -1
- package/dist/{default-registry-Dhrc__SE.d.ts → default-registry-ZhqsTr4K.d.ts} +2 -2
- package/dist/diagnose.d.ts +3 -3
- package/dist/diagnose.js +1 -1
- package/dist/{gepa-bxuDoaO9.d.ts → gepa-DeyPTlvx.d.ts} +1 -1
- package/dist/hosted/index.d.ts +4 -4
- package/dist/{index-W96macmS.d.ts → index-B-bFgiAF.d.ts} +1 -1
- package/dist/index.d.ts +27 -23
- package/dist/index.js +26 -10
- package/dist/index.js.map +1 -1
- package/dist/{insight-report-C02J3q4T.d.ts → insight-report-k0sRTzKg.d.ts} +1 -1
- package/dist/{kind-factory-OgqQSvLi.d.ts → kind-factory-D0nk7AKV.d.ts} +1 -1
- package/dist/meta-eval/index.d.ts +2 -2
- package/dist/multishot/index.d.ts +2 -2
- package/dist/openapi.json +1 -1
- package/dist/{policy-edit-Dccm9tyA.d.ts → policy-edit-BDQzzsBU.d.ts} +2 -2
- package/dist/{pre-registration-BjGZf9YA.d.ts → pre-registration-mWG2w8d-.d.ts} +45 -12
- package/dist/product-benchmark/index.d.ts +104 -1
- package/dist/product-benchmark/index.js +15 -1
- package/dist/{provenance-BEITkFII.d.ts → provenance-BhJm32vN.d.ts} +3 -3
- package/dist/{release-report-B1tA6pKu.d.ts → release-report-BQ1Ziyu-.d.ts} +2 -2
- package/dist/reporting.d.ts +4 -4
- package/dist/{researcher-Ba2y1Foi.d.ts → researcher-B_ODTAJs.d.ts} +2 -2
- package/dist/rl.d.ts +8 -8
- package/dist/rl.js +2 -2
- package/dist/{rubric-predictive-validity-w7tun-q3.d.ts → rubric-predictive-validity-0MdjTt8R.d.ts} +1 -1
- package/dist/{run-campaign-RF3H6D4U.js → run-campaign-2L4WCJHR.js} +2 -2
- package/dist/{run-record-DEwidcqn.d.ts → run-record-MRdJ-Kq2.d.ts} +12 -1
- package/dist/{runtime-trajectory-OJDaTYHN.d.ts → runtime-trajectory-8w0_jmtR.d.ts} +1 -1
- package/dist/{semantic-concept-judge-J8xvjdc3.d.ts → semantic-concept-judge-D-IlH5v1.d.ts} +2 -2
- package/dist/{summary-report-C4uzRWh8.d.ts → summary-report-C0nnxOD8.d.ts} +1 -1
- package/dist/traces.d.ts +1 -1
- package/dist/traces.js +2 -2
- package/dist/{types-fWqEJm7h.d.ts → types-DFI_Z-ZL.d.ts} +18 -1
- package/dist/{types-BEzCBMQD.d.ts → types-Dz9cKF0g.d.ts} +1 -1
- package/dist/workflow/index.d.ts +4 -4
- package/dist/workflow/index.js +1 -1
- package/package.json +1 -1
- package/dist/chunk-52CCCXU3.js.map +0 -1
- package/dist/chunk-63MBSQTX.js +0 -350
- package/dist/chunk-63MBSQTX.js.map +0 -1
- package/dist/chunk-CMJSTXUR.js.map +0 -1
- package/dist/chunk-LSCBODPQ.js.map +0 -1
- /package/dist/{chunk-4LWD6GC7.js.map → chunk-6FIAJHCU.js.map} +0 -0
- /package/dist/{chunk-PMF5WIBX.js.map → chunk-7RBJANJD.js.map} +0 -0
- /package/dist/{chunk-BOETF6BU.js.map → chunk-B2TMQM62.js.map} +0 -0
- /package/dist/{chunk-QUCGGMYM.js.map → chunk-HV5PBTJF.js.map} +0 -0
- /package/dist/{chunk-JCUREYF5.js.map → chunk-U3IDYATS.js.map} +0 -0
- /package/dist/{run-campaign-RF3H6D4U.js.map → run-campaign-2L4WCJHR.js.map} +0 -0
package/dist/adapters/http.d.ts
CHANGED
|
@@ -1,5 +1,5 @@
|
|
|
1
|
-
import { S as Scenario, D as DispatchFn, b as DispatchContext } from '../types-
|
|
2
|
-
import '../run-record-
|
|
1
|
+
import { S as Scenario, D as DispatchFn, b as DispatchContext } from '../types-DFI_Z-ZL.js';
|
|
2
|
+
import '../run-record-MRdJ-Kq2.js';
|
|
3
3
|
import '@tangle-network/agent-interface';
|
|
4
4
|
import '../errors-CzMUYo7b.js';
|
|
5
5
|
import '../schema-m0gsnbt3.js';
|
|
@@ -1,5 +1,5 @@
|
|
|
1
|
-
import { S as Scenario, J as JudgeScore, D as DispatchFn, a as JudgeConfig } from '../types-
|
|
2
|
-
import '../run-record-
|
|
1
|
+
import { S as Scenario, J as JudgeScore, D as DispatchFn, a as JudgeConfig } from '../types-DFI_Z-ZL.js';
|
|
2
|
+
import '../run-record-MRdJ-Kq2.js';
|
|
3
3
|
import '@tangle-network/agent-interface';
|
|
4
4
|
import '../errors-CzMUYo7b.js';
|
|
5
5
|
import '../schema-m0gsnbt3.js';
|
package/dist/adapters/otel.d.ts
CHANGED
|
@@ -1,11 +1,11 @@
|
|
|
1
1
|
import { TraceSpanEvent, HostedClient } from '../hosted/index.js';
|
|
2
|
-
import '../types-
|
|
3
|
-
import '../run-record-
|
|
2
|
+
import '../types-DFI_Z-ZL.js';
|
|
3
|
+
import '../run-record-MRdJ-Kq2.js';
|
|
4
4
|
import '@tangle-network/agent-interface';
|
|
5
5
|
import '../errors-CzMUYo7b.js';
|
|
6
6
|
import '../schema-m0gsnbt3.js';
|
|
7
|
-
import '../insight-report-
|
|
8
|
-
import '../summary-report-
|
|
7
|
+
import '../insight-report-k0sRTzKg.js';
|
|
8
|
+
import '../summary-report-C0nnxOD8.js';
|
|
9
9
|
import '../failure-cluster-DH9Flgcf.js';
|
|
10
10
|
import '../store-BcFXE6LG.js';
|
|
11
11
|
import '../judge-calibration-0p2QcWNE.js';
|
package/dist/analyst/index.d.ts
CHANGED
|
@@ -1,22 +1,22 @@
|
|
|
1
1
|
import { M as MultiLayerVerifier, V as VerifyOptions, S as Severity } from '../multi-layer-verifier-DUZXrPDA.js';
|
|
2
2
|
import { c as RunCritic, a as RunTrace } from '../run-critic-CmMf05uV.js';
|
|
3
|
-
import { S as SemanticConceptJudgeOptions, a as SemanticConceptJudgeInput, B as BehavioralMetrics } from '../semantic-concept-judge-
|
|
4
|
-
export { C as CreateAnalystAiConfig, D as DEFAULT_TRACE_ANALYST_KINDS, b as DiffPolicy, F as FAILURE_MODE_KIND_SPEC, c as FINDING_SUBJECT_GRAMMAR_PROMPT, d as FINDING_SUBJECT_KINDS, e as FindingSubject, f as FindingSubjectKind, g as FindingSubjectStringSchema, h as FindingsDiff, i as FindingsStore, I as IMPROVEMENT_KIND_SPEC, K as KIND_EXPECTED_SUBJECTS, j as KNOWLEDGE_GAP_KIND_SPEC, k as KNOWLEDGE_POISONING_KIND_SPEC, P as PersistedFinding, l as SKILL_USAGE_ANALYST, m as SkillUsageAnalyst, n as SkillUsageRecord, o as SkillUsageReport, p as SkillUsageScanConfig, q as buildSkillUsageReport, r as createAnalystAi, s as defaultIsMaterial, t as diffFindings, u as emitSkillUsageFindings, v as parseFindingSubject, w as renderFindingSubject } from '../semantic-concept-judge-
|
|
3
|
+
import { S as SemanticConceptJudgeOptions, a as SemanticConceptJudgeInput, B as BehavioralMetrics } from '../semantic-concept-judge-D-IlH5v1.js';
|
|
4
|
+
export { C as CreateAnalystAiConfig, D as DEFAULT_TRACE_ANALYST_KINDS, b as DiffPolicy, F as FAILURE_MODE_KIND_SPEC, c as FINDING_SUBJECT_GRAMMAR_PROMPT, d as FINDING_SUBJECT_KINDS, e as FindingSubject, f as FindingSubjectKind, g as FindingSubjectStringSchema, h as FindingsDiff, i as FindingsStore, I as IMPROVEMENT_KIND_SPEC, K as KIND_EXPECTED_SUBJECTS, j as KNOWLEDGE_GAP_KIND_SPEC, k as KNOWLEDGE_POISONING_KIND_SPEC, P as PersistedFinding, l as SKILL_USAGE_ANALYST, m as SkillUsageAnalyst, n as SkillUsageRecord, o as SkillUsageReport, p as SkillUsageScanConfig, q as buildSkillUsageReport, r as createAnalystAi, s as defaultIsMaterial, t as diffFindings, u as emitSkillUsageFindings, v as parseFindingSubject, w as renderFindingSubject } from '../semantic-concept-judge-D-IlH5v1.js';
|
|
5
5
|
import { b as JudgeFn, a as JudgeInput } from '../types-C7DGg5ex.js';
|
|
6
|
-
import { a as Analyst, h as AnalystSeverity, A as AnalystFinding } from '../types-
|
|
7
|
-
export { b as AnalystContext, g as AnalystCost, i as AnalystInputKind, j as AnalystRequirements, f as AnalystRunEvent, e as AnalystRunInputs, d as AnalystRunResult, c as AnalystRunSummary, k as ChatCallOpts, C as ChatClient, l as ChatRequest, m as ChatResponse, n as ChatTransport, o as CliBridgeTransportOpts, p as CreateChatClientOpts, D as DirectProviderTransportOpts, E as EvidenceRef, M as MockTransportOpts, R as RouterTransportOpts, S as SandboxSdkTransportOpts, q as computeFindingId, r as createChatClient, s as makeFinding } from '../types-
|
|
6
|
+
import { a as Analyst, h as AnalystSeverity, A as AnalystFinding } from '../types-Dz9cKF0g.js';
|
|
7
|
+
export { b as AnalystContext, g as AnalystCost, i as AnalystInputKind, j as AnalystRequirements, f as AnalystRunEvent, e as AnalystRunInputs, d as AnalystRunResult, c as AnalystRunSummary, k as ChatCallOpts, C as ChatClient, l as ChatRequest, m as ChatResponse, n as ChatTransport, o as CliBridgeTransportOpts, p as CreateChatClientOpts, D as DirectProviderTransportOpts, E as EvidenceRef, M as MockTransportOpts, R as RouterTransportOpts, S as SandboxSdkTransportOpts, q as computeFindingId, r as createChatClient, s as makeFinding } from '../types-Dz9cKF0g.js';
|
|
8
8
|
import { TCloud } from '@tangle-network/tcloud';
|
|
9
9
|
import { T as TraceAnalysisStore } from '../store-C1YxJDEK.js';
|
|
10
|
-
export { a as AnalystHooks, A as AnalystRegistry, b as AnalystRegistryOptions, B as BudgetPolicy, D as DefaultAnalystRegistryOptions, R as RegistryRunOpts, c as buildDefaultAnalystRegistry } from '../default-registry-
|
|
11
|
-
export { A as ANALYST_SEVERITIES, C as CreateTraceAnalystKindOpts, R as RAW_FINDING_SCHEMA_PROMPT, a as RawAnalystFinding, b as RawAnalystFindingSchema, c as TraceAnalystGolden, T as TraceAnalystKindSpec, d as createTraceAnalystKind, p as parseRawFinding, r as renderPriorFindings } from '../kind-factory-
|
|
12
|
-
export { F as FindingToPolicyEditOptions, P as POLICY_EDIT_AXES, a as POLICY_EDIT_TARGET_SURFACES, b as PolicyEdit, c as PolicyEditAdmission, d as PolicyEditAdmissionOptions, e as PolicyEditAxis, f as PolicyEditChange, g as PolicyEditExpectedGain, h as PolicyEditGainDirection, i as PolicyEditGainUnit, j as PolicyEditInit, k as PolicyEditRisk, l as PolicyEditSchemaVersion, m as PolicyEditSource, n as PolicyEditTarget, o as PolicyEditTargetSurface, p as PolicyEditValidationError, q as admitPolicyEdit, r as applyPolicyEditToSurface, s as computePolicyEditId, t as isPolicyEdit, u as makePolicyEdit, v as policyEditFromFinding, w as policyEditsFromFindings, x as scorePolicyEditReadiness, y as validatePolicyEdit } from '../policy-edit-
|
|
10
|
+
export { a as AnalystHooks, A as AnalystRegistry, b as AnalystRegistryOptions, B as BudgetPolicy, D as DefaultAnalystRegistryOptions, R as RegistryRunOpts, c as buildDefaultAnalystRegistry } from '../default-registry-ZhqsTr4K.js';
|
|
11
|
+
export { A as ANALYST_SEVERITIES, C as CreateTraceAnalystKindOpts, R as RAW_FINDING_SCHEMA_PROMPT, a as RawAnalystFinding, b as RawAnalystFindingSchema, c as TraceAnalystGolden, T as TraceAnalystKindSpec, d as createTraceAnalystKind, p as parseRawFinding, r as renderPriorFindings } from '../kind-factory-D0nk7AKV.js';
|
|
12
|
+
export { F as FindingToPolicyEditOptions, P as POLICY_EDIT_AXES, a as POLICY_EDIT_TARGET_SURFACES, b as PolicyEdit, c as PolicyEditAdmission, d as PolicyEditAdmissionOptions, e as PolicyEditAxis, f as PolicyEditChange, g as PolicyEditExpectedGain, h as PolicyEditGainDirection, i as PolicyEditGainUnit, j as PolicyEditInit, k as PolicyEditRisk, l as PolicyEditSchemaVersion, m as PolicyEditSource, n as PolicyEditTarget, o as PolicyEditTargetSurface, p as PolicyEditValidationError, q as admitPolicyEdit, r as applyPolicyEditToSurface, s as computePolicyEditId, t as isPolicyEdit, u as makePolicyEdit, v as policyEditFromFinding, w as policyEditsFromFindings, x as scorePolicyEditReadiness, y as validatePolicyEdit } from '../policy-edit-BDQzzsBU.js';
|
|
13
13
|
import { L as LlmClientOptions } from '../llm-client-Bj7g0rqu.js';
|
|
14
14
|
import { AxFunction } from '@ax-llm/ax';
|
|
15
15
|
import '../verdict-C9MlYujm.js';
|
|
16
16
|
import '../schema-m0gsnbt3.js';
|
|
17
17
|
import '../store-BcFXE6LG.js';
|
|
18
18
|
import 'zod';
|
|
19
|
-
import '../run-record-
|
|
19
|
+
import '../run-record-MRdJ-Kq2.js';
|
|
20
20
|
import '@tangle-network/agent-interface';
|
|
21
21
|
import '../errors-CzMUYo7b.js';
|
|
22
22
|
import '../raw-provider-sink-C46HDghv.js';
|
|
@@ -1,7 +1,7 @@
|
|
|
1
|
-
import { A as AnalystRegistry } from './default-registry-
|
|
1
|
+
import { A as AnalystRegistry } from './default-registry-ZhqsTr4K.js';
|
|
2
2
|
import { a as DatasetScenario } from './dataset-BbGkaN2I.js';
|
|
3
|
-
import { R as RunRecord } from './run-record-
|
|
4
|
-
import { I as InsightReport } from './insight-report-
|
|
3
|
+
import { R as RunRecord } from './run-record-MRdJ-Kq2.js';
|
|
4
|
+
import { I as InsightReport } from './insight-report-k0sRTzKg.js';
|
|
5
5
|
|
|
6
6
|
/**
|
|
7
7
|
* # `analyzeRuns()` — turn a set of agent runs into an actionable decision packet.
|
|
@@ -1,9 +1,9 @@
|
|
|
1
1
|
import { c as CalibrationReport } from '../calibration-BPmzuVPk.js';
|
|
2
2
|
import { O as OffPolicyEstimate, a as OffPolicyOptions, b as OffPolicyTrajectory } from '../off-policy-DiwuKKg7.js';
|
|
3
|
-
import { d as CodeAgentSessionSource, a as CodeAgentSessionIntakeOptions, c as CodeAgentSessionMetrics, C as CodeAgentSessionDiagnostic } from '../code-agent-session-
|
|
4
|
-
import { R as RunRecord, b as RunSplitTag } from '../run-record-
|
|
3
|
+
import { d as CodeAgentSessionSource, a as CodeAgentSessionIntakeOptions, c as CodeAgentSessionMetrics, C as CodeAgentSessionDiagnostic } from '../code-agent-session-Ce-9u7YM.js';
|
|
4
|
+
import { R as RunRecord, b as RunSplitTag } from '../run-record-MRdJ-Kq2.js';
|
|
5
5
|
import { T as TraceStore } from '../store-BcFXE6LG.js';
|
|
6
|
-
import { R as RuntimeTrajectoryRecord, P as ProjectRuntimeTrajectoryEvidenceOptions, a as RuntimeTrajectoryEvidenceProjection } from '../runtime-trajectory-
|
|
6
|
+
import { R as RuntimeTrajectoryRecord, P as ProjectRuntimeTrajectoryEvidenceOptions, a as RuntimeTrajectoryEvidenceProjection } from '../runtime-trajectory-8w0_jmtR.js';
|
|
7
7
|
import '../schema-m0gsnbt3.js';
|
|
8
8
|
import '../outcome-store-rnXLEqSn.js';
|
|
9
9
|
import '@tangle-network/agent-interface';
|
|
@@ -1,5 +1,5 @@
|
|
|
1
|
-
export { B as BENCHMARK_SPLIT_SEED, a as BenchmarkAdapter, b as BenchmarkDatasetItem, c as BenchmarkEvaluation, d as deterministicSplit, e as routing } from '../index-
|
|
2
|
-
import '../run-record-
|
|
1
|
+
export { B as BENCHMARK_SPLIT_SEED, a as BenchmarkAdapter, b as BenchmarkDatasetItem, c as BenchmarkEvaluation, d as deterministicSplit, e as routing } from '../index-B-bFgiAF.js';
|
|
2
|
+
import '../run-record-MRdJ-Kq2.js';
|
|
3
3
|
import '@tangle-network/agent-interface';
|
|
4
4
|
import '../errors-CzMUYo7b.js';
|
|
5
5
|
import '../schema-m0gsnbt3.js';
|
package/dist/campaign/index.d.ts
CHANGED
|
@@ -1,20 +1,21 @@
|
|
|
1
|
-
import { S as SignedManifest, B as BackendIntegrityReport, C as CompletionRequirement, R as RuntimeEventLike, a as CompletionVerdict, P as ProducedState, b as CorrectnessChecker } from '../pre-registration-
|
|
2
|
-
export { L as LlmJudgeDimension, c as LlmJudgeOptions, l as llmJudge } from '../pre-registration-
|
|
1
|
+
import { S as SignedManifest, B as BackendIntegrityReport, C as CompletionRequirement, R as RuntimeEventLike, a as CompletionVerdict, P as ProducedState, b as CorrectnessChecker } from '../pre-registration-mWG2w8d-.js';
|
|
2
|
+
export { L as LlmJudgeDimension, c as LlmJudgeOptions, l as llmJudge } from '../pre-registration-mWG2w8d-.js';
|
|
3
3
|
import { A as AnalyzeTracesOptions, a as AnalyzeTracesInput, b as AnalyzeTracesResult } from '../analyst-C8HHvfJp.js';
|
|
4
|
-
import { S as Scenario, M as MutableSurface, b as DispatchContext, a as JudgeConfig, e as GenerationRecord, g as Gate, J as JudgeScore, L as LabeledScenarioStore, s as LabeledScenarioWrite, t as LabeledScenarioSampleArgs, u as LabeledScenarioRecord, v as LabelTrust, f as SurfaceProposer, w as ProposedCandidate, x as ProposeContext, y as LabeledScenarioSource, C as CampaignResult, o as CodeSurface } from '../types-
|
|
5
|
-
export { k as CampaignAggregates, l as CampaignArtifactWriter, m as CampaignCellResult, n as CampaignCostMeter, z as CampaignTokenUsage, d as CampaignTraceWriter, D as DispatchFn, h as GateContext, j as GateDecision, G as GateResult, p as GenerationCandidate, A as JudgeAggregate, c as JudgeDimension, i as Mutator, O as OptimizationProposer, q as OptimizerConfig, P as ParetoParent, R as RedactionStatus, B as ScenarioAggregate, r as SessionScript, T as TraceSpan, E as isProposedCandidate, F as labelTrustRank } from '../types-
|
|
6
|
-
import { e as CampaignRunPlan, P as PlanCampaignRunOptions, C as CampaignStorage, R as RunCampaignOptions, c as RunImprovementLoopOptions } from '../gepa-
|
|
7
|
-
export { h as CampaignRunPlanCell, j as GepaProposerConstraints, G as GepaProposerOptions, O as OpenAutoPrOptions, k as OpenAutoPrResult, b as RunImprovementLoopResult, a as RunOptimizationOptions, l as RunOptimizationResult, m as countSentenceEdits, n as defaultRenderDiff, o as extractH2Sections, f as fsCampaignStorage, g as gepaProposer, i as inMemoryCampaignStorage, p as openAutoPr, q as planCampaignRun, r as runCampaign, d as runImprovementLoop, s as runOptimization, t as surfaceHash } from '../gepa-
|
|
8
|
-
export { A as AxisEvidence, a as AxisVerdict, B as BuildEvidenceVectorOptions, k as BuildLoopProvenanceArgs, D as DefaultProductionGateOptions, l as EmitLoopProvenanceArgs, m as EmitLoopProvenanceResult, E as EvidenceVector, b as EvolutionaryProposerOptions, H as HeldOutGateOptions, n as LoopProvenanceBackend, o as LoopProvenanceCandidate, L as LoopProvenanceRecord, O as ObjectiveSource, P as ParetoSignificanceGateOptions, c as PromotionObjective, d as PromotionPolicy, R as RunEvalOptions, e as buildEvidenceVector, q as buildLoopProvenanceRecord, f as composeGate, g as defaultProductionGate, s as emitLoopProvenance, h as evolutionaryProposer, i as heldOutGate, t as loopProvenanceSpans, p as paretoPolicy, j as paretoSignificanceGate, u as provenanceRecordPath, v as provenanceSpansPath, r as runEval, w as surfaceContentHash } from '../provenance-
|
|
4
|
+
import { S as Scenario, M as MutableSurface, b as DispatchContext, a as JudgeConfig, e as GenerationRecord, g as Gate, J as JudgeScore, L as LabeledScenarioStore, s as LabeledScenarioWrite, t as LabeledScenarioSampleArgs, u as LabeledScenarioRecord, v as LabelTrust, f as SurfaceProposer, w as ProposedCandidate, x as ProposeContext, y as LabeledScenarioSource, C as CampaignResult, o as CodeSurface } from '../types-DFI_Z-ZL.js';
|
|
5
|
+
export { k as CampaignAggregates, l as CampaignArtifactWriter, m as CampaignCellResult, n as CampaignCostMeter, z as CampaignTokenUsage, d as CampaignTraceWriter, D as DispatchFn, h as GateContext, j as GateDecision, G as GateResult, p as GenerationCandidate, A as JudgeAggregate, c as JudgeDimension, i as Mutator, O as OptimizationProposer, q as OptimizerConfig, P as ParetoParent, R as RedactionStatus, B as ScenarioAggregate, r as SessionScript, T as TraceSpan, E as isProposedCandidate, F as labelTrustRank } from '../types-DFI_Z-ZL.js';
|
|
6
|
+
import { e as CampaignRunPlan, P as PlanCampaignRunOptions, C as CampaignStorage, R as RunCampaignOptions, c as RunImprovementLoopOptions } from '../gepa-DeyPTlvx.js';
|
|
7
|
+
export { h as CampaignRunPlanCell, j as GepaProposerConstraints, G as GepaProposerOptions, O as OpenAutoPrOptions, k as OpenAutoPrResult, b as RunImprovementLoopResult, a as RunOptimizationOptions, l as RunOptimizationResult, m as countSentenceEdits, n as defaultRenderDiff, o as extractH2Sections, f as fsCampaignStorage, g as gepaProposer, i as inMemoryCampaignStorage, p as openAutoPr, q as planCampaignRun, r as runCampaign, d as runImprovementLoop, s as runOptimization, t as surfaceHash } from '../gepa-DeyPTlvx.js';
|
|
8
|
+
export { A as AxisEvidence, a as AxisVerdict, B as BuildEvidenceVectorOptions, k as BuildLoopProvenanceArgs, D as DefaultProductionGateOptions, l as EmitLoopProvenanceArgs, m as EmitLoopProvenanceResult, E as EvidenceVector, b as EvolutionaryProposerOptions, H as HeldOutGateOptions, n as LoopProvenanceBackend, o as LoopProvenanceCandidate, L as LoopProvenanceRecord, O as ObjectiveSource, P as ParetoSignificanceGateOptions, c as PromotionObjective, d as PromotionPolicy, R as RunEvalOptions, e as buildEvidenceVector, q as buildLoopProvenanceRecord, f as composeGate, g as defaultProductionGate, s as emitLoopProvenance, h as evolutionaryProposer, i as heldOutGate, t as loopProvenanceSpans, p as paretoPolicy, j as paretoSignificanceGate, u as provenanceRecordPath, v as provenanceSpansPath, r as runEval, w as surfaceContentHash } from '../provenance-BhJm32vN.js';
|
|
9
9
|
import { E as EProcessState, a as PairedBootstrapResult } from '../statistics-xP-cWc5k.js';
|
|
10
10
|
import { L as LlmClientOptions } from '../llm-client-Bj7g0rqu.js';
|
|
11
11
|
import { AgentProfile } from '@tangle-network/agent-interface';
|
|
12
12
|
import { A as AgentEvalError } from '../errors-CzMUYo7b.js';
|
|
13
|
-
import { b as RunSplitTag, R as RunRecord } from '../run-record-
|
|
14
|
-
import { b as PolicyEdit, F as FindingToPolicyEditOptions, d as PolicyEditAdmissionOptions, c as PolicyEditAdmission } from '../policy-edit-
|
|
15
|
-
import { T as TraceAnalystKindSpec } from '../kind-factory-
|
|
16
|
-
import { A as AnalystFinding } from '../types-
|
|
13
|
+
import { b as RunSplitTag, R as RunRecord } from '../run-record-MRdJ-Kq2.js';
|
|
14
|
+
import { b as PolicyEdit, F as FindingToPolicyEditOptions, d as PolicyEditAdmissionOptions, c as PolicyEditAdmission } from '../policy-edit-BDQzzsBU.js';
|
|
15
|
+
import { T as TraceAnalystKindSpec } from '../kind-factory-D0nk7AKV.js';
|
|
16
|
+
import { A as AnalystFinding } from '../types-Dz9cKF0g.js';
|
|
17
17
|
import '@tangle-network/tcloud';
|
|
18
|
+
import '../raw-provider-sink-C46HDghv.js';
|
|
18
19
|
import '../verdict-C9MlYujm.js';
|
|
19
20
|
import '@ax-llm/ax';
|
|
20
21
|
import '../store-C1YxJDEK.js';
|
|
@@ -24,12 +25,11 @@ import '../store-BcFXE6LG.js';
|
|
|
24
25
|
import '../schema-m0gsnbt3.js';
|
|
25
26
|
import '../pareto-E-pembql.js';
|
|
26
27
|
import '../hosted/index.js';
|
|
27
|
-
import '../insight-report-
|
|
28
|
-
import '../summary-report-
|
|
28
|
+
import '../insight-report-k0sRTzKg.js';
|
|
29
|
+
import '../summary-report-C0nnxOD8.js';
|
|
29
30
|
import '../failure-cluster-DH9Flgcf.js';
|
|
30
31
|
import '../judge-calibration-0p2QcWNE.js';
|
|
31
32
|
import '../types-C7DGg5ex.js';
|
|
32
|
-
import '../raw-provider-sink-C46HDghv.js';
|
|
33
33
|
import 'zod';
|
|
34
34
|
|
|
35
35
|
/**
|
package/dist/campaign/index.js
CHANGED
|
@@ -10,8 +10,9 @@ import {
|
|
|
10
10
|
paretoPolicy,
|
|
11
11
|
paretoSignificanceGate,
|
|
12
12
|
runEval
|
|
13
|
-
} from "../chunk-
|
|
13
|
+
} from "../chunk-HV5PBTJF.js";
|
|
14
14
|
import {
|
|
15
|
+
HARNESS_NATIVE_MODEL,
|
|
15
16
|
agentProfileHash,
|
|
16
17
|
agentProfileId,
|
|
17
18
|
agentProfileModelId,
|
|
@@ -19,7 +20,7 @@ import {
|
|
|
19
20
|
harnessAxisOf,
|
|
20
21
|
llmJudge,
|
|
21
22
|
verifyCompletion
|
|
22
|
-
} from "../chunk-
|
|
23
|
+
} from "../chunk-IXOV77YF.js";
|
|
23
24
|
import {
|
|
24
25
|
buildLoopProvenanceRecord,
|
|
25
26
|
campaignBreakdown,
|
|
@@ -41,7 +42,7 @@ import {
|
|
|
41
42
|
runOptimization,
|
|
42
43
|
surfaceContentHash,
|
|
43
44
|
surfaceHash
|
|
44
|
-
} from "../chunk-
|
|
45
|
+
} from "../chunk-NTVWIH24.js";
|
|
45
46
|
import {
|
|
46
47
|
assertRealBackend,
|
|
47
48
|
contentHash,
|
|
@@ -52,7 +53,7 @@ import {
|
|
|
52
53
|
runCampaign,
|
|
53
54
|
summarizeBackendIntegrity,
|
|
54
55
|
tangleTracesRoot
|
|
55
|
-
} from "../chunk-
|
|
56
|
+
} from "../chunk-XKA6ZGEY.js";
|
|
56
57
|
import {
|
|
57
58
|
estimateCost,
|
|
58
59
|
isModelPriced
|
|
@@ -87,8 +88,9 @@ import {
|
|
|
87
88
|
} from "../chunk-YBIGNSCZ.js";
|
|
88
89
|
import "../chunk-PC4UYEBM.js";
|
|
89
90
|
import {
|
|
91
|
+
modelHasSnapshot,
|
|
90
92
|
validateRunRecord
|
|
91
|
-
} from "../chunk-
|
|
93
|
+
} from "../chunk-2NSLDY4B.js";
|
|
92
94
|
import {
|
|
93
95
|
buildAgentProfileCell
|
|
94
96
|
} from "../chunk-ABOIVNXL.js";
|
|
@@ -2031,10 +2033,25 @@ function cellComposite(cell) {
|
|
|
2031
2033
|
const composites = Object.values(cell.judgeScores).map((s) => s.composite);
|
|
2032
2034
|
return composites.length === 0 ? 0 : mean2(composites);
|
|
2033
2035
|
}
|
|
2036
|
+
function requireResolvedModel(cell, profileId) {
|
|
2037
|
+
const resolved = cell.resolvedModel?.trim();
|
|
2038
|
+
if (!resolved) {
|
|
2039
|
+
throw new ProfileMatrixError(
|
|
2040
|
+
`profile '${profileId}' declared the '${HARNESS_NATIVE_MODEL}' runtime-resolved model but its dispatch reported no resolved model for cell '${cell.cellId}' \u2014 report it via ctx.cost.observeModel(<id>) so the RunRecord pins the real model (never records '${HARNESS_NATIVE_MODEL}')`
|
|
2041
|
+
);
|
|
2042
|
+
}
|
|
2043
|
+
if (!modelHasSnapshot(resolved)) {
|
|
2044
|
+
throw new ProfileMatrixError(
|
|
2045
|
+
`profile '${profileId}' resolved to model '${resolved}' for cell '${cell.cellId}', which lacks a snapshot version \u2014 pin it (name@YYYY-MM-DD or name-YYYYMMDD) before reporting it via ctx.cost.observeModel`
|
|
2046
|
+
);
|
|
2047
|
+
}
|
|
2048
|
+
return resolved;
|
|
2049
|
+
}
|
|
2034
2050
|
function buildRunRecord(args) {
|
|
2035
2051
|
const { cell, profile, profileHash, configHash, experimentId, splitTag, commitSha, matrixId } = args;
|
|
2036
2052
|
const profileId = agentProfileId(profile);
|
|
2037
|
-
const
|
|
2053
|
+
const declaredModel = agentProfileModelId(profile);
|
|
2054
|
+
const model = declaredModel === HARNESS_NATIVE_MODEL ? requireResolvedModel(cell, profileId) : declaredModel;
|
|
2038
2055
|
const composite = cellComposite(cell);
|
|
2039
2056
|
const raw = { composite };
|
|
2040
2057
|
const perJudge = {};
|
|
@@ -2120,7 +2137,8 @@ async function runProfileMatrix(opts) {
|
|
|
2120
2137
|
for (const profile of opts.profiles) {
|
|
2121
2138
|
const profileHash = agentProfileHash(profile);
|
|
2122
2139
|
const profileId = agentProfileId(profile);
|
|
2123
|
-
const
|
|
2140
|
+
const declaredModel = agentProfileModelId(profile);
|
|
2141
|
+
const model = declaredModel === HARNESS_NATIVE_MODEL ? `${HARNESS_NATIVE_MODEL}@runtime-resolved` : declaredModel;
|
|
2124
2142
|
try {
|
|
2125
2143
|
validateRunRecord({
|
|
2126
2144
|
runId: `${matrixId}:${profileId}:probe`,
|
|
@@ -2149,7 +2167,7 @@ async function runProfileMatrix(opts) {
|
|
|
2149
2167
|
for (const profile of opts.profiles) {
|
|
2150
2168
|
const profileHash = agentProfileHash(profile);
|
|
2151
2169
|
const profileId = agentProfileId(profile);
|
|
2152
|
-
const
|
|
2170
|
+
const declaredModel = agentProfileModelId(profile);
|
|
2153
2171
|
const configHash = sha({
|
|
2154
2172
|
profile: profileHash,
|
|
2155
2173
|
judges: (opts.judges ?? []).map((j) => j.name),
|
|
@@ -2173,14 +2191,16 @@ async function runProfileMatrix(opts) {
|
|
|
2173
2191
|
runDir: join3(opts.runDir, sanitize(profileId))
|
|
2174
2192
|
});
|
|
2175
2193
|
const axis = harnessAxisOf(profile);
|
|
2176
|
-
const
|
|
2194
|
+
const buildCellIdentity = (cellModel) => buildAgentProfileCell({
|
|
2177
2195
|
profileId,
|
|
2178
2196
|
sourceProfile: { kind: "agent-interface-profile", hash: profileHash },
|
|
2179
|
-
model,
|
|
2197
|
+
model: cellModel,
|
|
2180
2198
|
...axis ? { harness: { id: axis.harness } } : {}
|
|
2181
2199
|
});
|
|
2200
|
+
const sharedCellIdentity = declaredModel === HARNESS_NATIVE_MODEL ? void 0 : await buildCellIdentity(declaredModel);
|
|
2182
2201
|
const profileRecords = [];
|
|
2183
2202
|
for (const cell of campaign.cells) {
|
|
2203
|
+
const agentProfileCell = sharedCellIdentity ?? await buildCellIdentity(requireResolvedModel(cell, profileId));
|
|
2184
2204
|
const record = buildRunRecord({
|
|
2185
2205
|
cell,
|
|
2186
2206
|
profile,
|
|
@@ -2206,7 +2226,10 @@ async function runProfileMatrix(opts) {
|
|
|
2206
2226
|
byProfile[profileId] = {
|
|
2207
2227
|
profileId,
|
|
2208
2228
|
profileHash,
|
|
2209
|
-
model,
|
|
2229
|
+
// The declared model, unless it snapped to the sentinel — then the
|
|
2230
|
+
// resolved model the cells actually ran on (all cells of a profile share
|
|
2231
|
+
// one harness, so the first record's model is representative).
|
|
2232
|
+
model: declaredModel === HARNESS_NATIVE_MODEL ? profileRecords[0]?.model ?? declaredModel : declaredModel,
|
|
2210
2233
|
records: profileRecords.length,
|
|
2211
2234
|
meanComposite: mean2(profileRecords.map(compositeOf)),
|
|
2212
2235
|
totalCostUsd: pricedTotalCostUsd,
|