@tangle-network/agent-eval 0.103.0 → 0.103.1
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/dist/adapters/http.d.ts +2 -2
- package/dist/adapters/langchain.d.ts +2 -2
- package/dist/adapters/otel.d.ts +5 -5
- package/dist/analyst/index.d.ts +9 -9
- package/dist/analyst/index.js +3 -3
- package/dist/{analyze-runs-Cd-A_K4l.d.ts → analyze-runs-rF2eGxZ_.d.ts} +3 -3
- package/dist/belief-state/index.d.ts +3 -3
- package/dist/belief-state/index.js +1 -1
- package/dist/benchmarks/index.d.ts +2 -2
- package/dist/builder-eval/index.js +2 -2
- package/dist/campaign/index.d.ts +42 -15
- package/dist/campaign/index.js +9 -9
- package/dist/campaign/index.js.map +1 -1
- package/dist/{chunk-NTVWIH24.js → chunk-2B4HPFXN.js} +3 -3
- package/dist/chunk-2B4HPFXN.js.map +1 -0
- package/dist/{chunk-7RBJANJD.js → chunk-3722PKPB.js} +2 -2
- package/dist/{chunk-HV5PBTJF.js → chunk-66T25VWE.js} +5 -5
- package/dist/chunk-66T25VWE.js.map +1 -0
- package/dist/{chunk-6FIAJHCU.js → chunk-6YVAN6R3.js} +4 -4
- package/dist/{chunk-ABOIVNXL.js → chunk-AGYDMORK.js} +1 -1
- package/dist/chunk-AGYDMORK.js.map +1 -0
- package/dist/{chunk-B2TMQM62.js → chunk-BHCFJGL4.js} +2 -2
- package/dist/{chunk-IH7LYRHL.js → chunk-BXZVBX3D.js} +3 -3
- package/dist/{chunk-6GT4NI4V.js → chunk-CLS3374R.js} +2 -2
- package/dist/{chunk-HRGUJTER.js → chunk-HYGRFL7C.js} +1 -1
- package/dist/{chunk-HRGUJTER.js.map → chunk-HYGRFL7C.js.map} +1 -1
- package/dist/{chunk-2NSLDY4B.js → chunk-J5MUFXEY.js} +2 -2
- package/dist/{chunk-UFSG7ACU.js → chunk-JZXGWLK5.js} +1 -1
- package/dist/chunk-JZXGWLK5.js.map +1 -0
- package/dist/{chunk-IN3SHQML.js → chunk-LMSQ6EFA.js} +2 -2
- package/dist/{chunk-AIGWQEME.js → chunk-LOZOZYHU.js} +2 -2
- package/dist/{chunk-NF7OZ4J7.js → chunk-QCBB6ZIU.js} +2 -2
- package/dist/{chunk-JU6ZX3CX.js → chunk-SMCACT4Z.js} +2 -2
- package/dist/{chunk-U3IDYATS.js → chunk-TCPP4Z5Q.js} +5 -5
- package/dist/{chunk-VDGPPGE3.js → chunk-UMEAR2FI.js} +2 -2
- package/dist/{chunk-VDGPPGE3.js.map → chunk-UMEAR2FI.js.map} +1 -1
- package/dist/{chunk-XKA6ZGEY.js → chunk-VNM52AGA.js} +2 -2
- package/dist/chunk-VNM52AGA.js.map +1 -0
- package/dist/{chunk-IXOV77YF.js → chunk-YFYIOSNV.js} +4 -4
- package/dist/{code-agent-session-Ce-9u7YM.d.ts → code-agent-session-Cen-qD2y.d.ts} +1 -1
- package/dist/contract/index.d.ts +18 -18
- package/dist/contract/index.js +10 -10
- package/dist/{control-C8RmK9H4.d.ts → control-KofK3gfG.d.ts} +1 -1
- package/dist/control.d.ts +2 -2
- package/dist/control.js +3 -3
- package/dist/{corpus-CiSzzLa5.d.ts → corpus-CysLCiK4.d.ts} +1 -1
- package/dist/{default-registry-ZhqsTr4K.d.ts → default-registry-Ez4cxuZ4.d.ts} +2 -2
- package/dist/diagnose.d.ts +3 -3
- package/dist/diagnose.js +3 -3
- package/dist/{gepa-DeyPTlvx.d.ts → gepa-COlCAkHN.d.ts} +22 -1
- package/dist/governance/index.d.ts +1 -1
- package/dist/hosted/index.d.ts +5 -5
- package/dist/{index-B-bFgiAF.d.ts → index-unCSYRJJ.d.ts} +1 -1
- package/dist/index.d.ts +31 -27
- package/dist/index.js +16 -16
- package/dist/index.js.map +1 -1
- package/dist/{insight-report-k0sRTzKg.d.ts → insight-report-DumCfEur.d.ts} +2 -2
- package/dist/{judge-calibration-0p2QcWNE.d.ts → judge-calibration-7C-IDmKr.d.ts} +3 -0
- package/dist/{kind-factory-D0nk7AKV.d.ts → kind-factory-BLHxwX71.d.ts} +1 -1
- package/dist/meta-eval/index.d.ts +4 -4
- package/dist/meta-eval/index.js +3 -3
- package/dist/{multi-layer-verifier-DUZXrPDA.d.ts → multi-layer-verifier-CI4jdX-q.d.ts} +3 -0
- package/dist/multishot/index.d.ts +2 -2
- package/dist/openapi.json +1 -1
- package/dist/pipelines/index.d.ts +1 -1
- package/dist/pipelines/index.js +3 -3
- package/dist/{policy-edit-BDQzzsBU.d.ts → policy-edit-nUhuFtLF.d.ts} +2 -2
- package/dist/{pre-registration-mWG2w8d-.d.ts → pre-registration-CUOSGAZK.d.ts} +3 -3
- package/dist/product-benchmark/index.d.ts +1 -1
- package/dist/{provenance-BhJm32vN.d.ts → provenance-B2VsP0jP.d.ts} +20 -4
- package/dist/{query-B7GGjRox.d.ts → query-0aTmbmQe.d.ts} +1 -0
- package/dist/{release-report-BQ1Ziyu-.d.ts → release-report-DkaCZ9k4.d.ts} +5 -2
- package/dist/reporting.d.ts +6 -6
- package/dist/reporting.js +4 -4
- package/dist/{researcher-B_ODTAJs.d.ts → researcher-SDAezfML.d.ts} +2 -2
- package/dist/rl.d.ts +9 -9
- package/dist/rl.js +7 -7
- package/dist/{rubric-predictive-validity-0MdjTt8R.d.ts → rubric-predictive-validity-ZGIlJNce.d.ts} +1 -1
- package/dist/{run-campaign-2L4WCJHR.js → run-campaign-DAHKO5CT.js} +3 -3
- package/dist/{run-record-MRdJ-Kq2.d.ts → run-record-CPfd1ARZ.d.ts} +3 -0
- package/dist/{runtime-trajectory-8w0_jmtR.d.ts → runtime-trajectory-DZ8ei-Jo.d.ts} +1 -1
- package/dist/{semantic-concept-judge-D-IlH5v1.d.ts → semantic-concept-judge-C6mDEBIo.d.ts} +3 -3
- package/dist/{statistics-xP-cWc5k.d.ts → statistics-D88peojY.d.ts} +1 -1
- package/dist/{summary-report-C0nnxOD8.d.ts → summary-report-Fc_YFJat.d.ts} +1 -1
- package/dist/traces.d.ts +2 -2
- package/dist/traces.js +4 -4
- package/dist/{types-DFI_Z-ZL.d.ts → types-Bihq6-a3.d.ts} +1 -1
- package/dist/{types-Dz9cKF0g.d.ts → types-DA9yj-Jd.d.ts} +1 -1
- package/dist/workflow/index.d.ts +7 -7
- package/dist/workflow/index.js +3 -3
- package/package.json +1 -1
- package/dist/chunk-ABOIVNXL.js.map +0 -1
- package/dist/chunk-HV5PBTJF.js.map +0 -1
- package/dist/chunk-NTVWIH24.js.map +0 -1
- package/dist/chunk-UFSG7ACU.js.map +0 -1
- package/dist/chunk-XKA6ZGEY.js.map +0 -1
- /package/dist/{chunk-7RBJANJD.js.map → chunk-3722PKPB.js.map} +0 -0
- /package/dist/{chunk-6FIAJHCU.js.map → chunk-6YVAN6R3.js.map} +0 -0
- /package/dist/{chunk-B2TMQM62.js.map → chunk-BHCFJGL4.js.map} +0 -0
- /package/dist/{chunk-IH7LYRHL.js.map → chunk-BXZVBX3D.js.map} +0 -0
- /package/dist/{chunk-6GT4NI4V.js.map → chunk-CLS3374R.js.map} +0 -0
- /package/dist/{chunk-2NSLDY4B.js.map → chunk-J5MUFXEY.js.map} +0 -0
- /package/dist/{chunk-IN3SHQML.js.map → chunk-LMSQ6EFA.js.map} +0 -0
- /package/dist/{chunk-AIGWQEME.js.map → chunk-LOZOZYHU.js.map} +0 -0
- /package/dist/{chunk-NF7OZ4J7.js.map → chunk-QCBB6ZIU.js.map} +0 -0
- /package/dist/{chunk-JU6ZX3CX.js.map → chunk-SMCACT4Z.js.map} +0 -0
- /package/dist/{chunk-U3IDYATS.js.map → chunk-TCPP4Z5Q.js.map} +0 -0
- /package/dist/{chunk-IXOV77YF.js.map → chunk-YFYIOSNV.js.map} +0 -0
- /package/dist/{run-campaign-2L4WCJHR.js.map → run-campaign-DAHKO5CT.js.map} +0 -0
|
@@ -1,5 +1,5 @@
|
|
|
1
|
-
import { G as GainDistributionBin, P as ParetoFigureSpec } from './summary-report-
|
|
2
|
-
import { C as ContinuousAgreement } from './judge-calibration-
|
|
1
|
+
import { G as GainDistributionBin, P as ParetoFigureSpec } from './summary-report-Fc_YFJat.js';
|
|
2
|
+
import { C as ContinuousAgreement } from './judge-calibration-7C-IDmKr.js';
|
|
3
3
|
|
|
4
4
|
/**
|
|
5
5
|
* # InsightReport — the rigorous decision packet for any set of agent runs.
|
|
@@ -45,6 +45,9 @@ interface CalibrationResult {
|
|
|
45
45
|
delta: number;
|
|
46
46
|
}>;
|
|
47
47
|
}
|
|
48
|
+
/**
|
|
49
|
+
* Measure judge quality against human gold labels: computes Cohen's κ, Pearson correlation, and MAE over matched item ids.
|
|
50
|
+
*/
|
|
48
51
|
declare function calibrateJudge(golden: GoldenItem[], candidate: CandidateScore[]): CalibrationResult;
|
|
49
52
|
interface PositionalBiasResult {
|
|
50
53
|
/**
|
|
@@ -1,7 +1,7 @@
|
|
|
1
1
|
import { AxAIService, AxFunction } from '@ax-llm/ax';
|
|
2
2
|
import { T as TraceAnalysisStore } from './store-C1YxJDEK.js';
|
|
3
3
|
import { z } from 'zod';
|
|
4
|
-
import { g as AnalystCost, b as AnalystContext, a as Analyst } from './types-
|
|
4
|
+
import { g as AnalystCost, b as AnalystContext, a as Analyst } from './types-DA9yj-Jd.js';
|
|
5
5
|
|
|
6
6
|
/**
|
|
7
7
|
* Typed Ax output for analyst findings.
|
|
@@ -1,12 +1,12 @@
|
|
|
1
1
|
export { C as CalibrationBin, a as CalibrationOptions, b as CalibrationPair, c as CalibrationReport, d as CorrelationResult, e as CorrelationStudyOptions, f as CorrelationStudyResult, E as EvalMetricSpec, O as OutcomePair, g as calibrationCurve, h as calibrationFromPairs, i as correlationStudy } from '../calibration-BPmzuVPk.js';
|
|
2
2
|
export { D as DeploymentOutcome, F as FileSystemOutcomeStore, a as FileSystemOutcomeStoreOptions, I as InMemoryOutcomeStore, O as OutcomeFilter, b as OutcomeStore } from '../outcome-store-rnXLEqSn.js';
|
|
3
|
-
export { R as RubricOutcomePair, a as RubricPredictiveValidityInput, b as RubricPredictiveValidityReport, c as RubricRanking, r as rubricPredictiveValidity } from '../rubric-predictive-validity-
|
|
4
|
-
import { C as ContinuousAgreement, a as CalibrationResult, b as ContinuousCalibrationResult, c as CandidateScore, G as GoldenItem } from '../judge-calibration-
|
|
3
|
+
export { R as RubricOutcomePair, a as RubricPredictiveValidityInput, b as RubricPredictiveValidityReport, c as RubricRanking, r as rubricPredictiveValidity } from '../rubric-predictive-validity-ZGIlJNce.js';
|
|
4
|
+
import { C as ContinuousAgreement, a as CalibrationResult, b as ContinuousCalibrationResult, c as CandidateScore, G as GoldenItem } from '../judge-calibration-7C-IDmKr.js';
|
|
5
5
|
import { S as SeriesConvergenceOptions, a as SeriesConvergenceResult } from '../series-convergence-D5OWMBg6.js';
|
|
6
|
-
import { C as CorpusAgreementReport } from '../statistics-
|
|
6
|
+
import { C as CorpusAgreementReport } from '../statistics-D88peojY.js';
|
|
7
7
|
import '../store-BcFXE6LG.js';
|
|
8
8
|
import '../schema-m0gsnbt3.js';
|
|
9
|
-
import '../run-record-
|
|
9
|
+
import '../run-record-CPfd1ARZ.js';
|
|
10
10
|
import '@tangle-network/agent-interface';
|
|
11
11
|
import '../errors-CzMUYo7b.js';
|
|
12
12
|
import '../types-C7DGg5ex.js';
|
package/dist/meta-eval/index.js
CHANGED
|
@@ -11,15 +11,15 @@ import {
|
|
|
11
11
|
} from "../chunk-3RF76KTD.js";
|
|
12
12
|
import {
|
|
13
13
|
rubricPredictiveValidity
|
|
14
|
-
} from "../chunk-
|
|
14
|
+
} from "../chunk-QCBB6ZIU.js";
|
|
15
15
|
import {
|
|
16
16
|
pearsonR,
|
|
17
17
|
spearmanR
|
|
18
|
-
} from "../chunk-
|
|
18
|
+
} from "../chunk-HYGRFL7C.js";
|
|
19
19
|
import {
|
|
20
20
|
aggregateLlm,
|
|
21
21
|
llmSpans
|
|
22
|
-
} from "../chunk-
|
|
22
|
+
} from "../chunk-JZXGWLK5.js";
|
|
23
23
|
import "../chunk-5BKGXME7.js";
|
|
24
24
|
import {
|
|
25
25
|
ValidationError
|
|
@@ -138,6 +138,9 @@ declare function gradeSemanticStatus(input: {
|
|
|
138
138
|
available: boolean;
|
|
139
139
|
threshold?: number;
|
|
140
140
|
}): LayerStatus;
|
|
141
|
+
/**
|
|
142
|
+
* Ordered DAG of verification layers with dependency-based skipping, per-layer findings, soft-fail semantics, and a blended composite score across all passed layers.
|
|
143
|
+
*/
|
|
141
144
|
declare class MultiLayerVerifier<Env = unknown> {
|
|
142
145
|
private readonly layers;
|
|
143
146
|
constructor(layers: Layer<Env>[]);
|
|
@@ -1,7 +1,7 @@
|
|
|
1
|
-
import { J as JudgeScore } from '../types-
|
|
1
|
+
import { J as JudgeScore } from '../types-Bihq6-a3.js';
|
|
2
2
|
import { AgentProfile } from '@tangle-network/agent-interface';
|
|
3
3
|
import { M as MatrixResult } from '../types-BUxNaJ8c.js';
|
|
4
|
-
import '../run-record-
|
|
4
|
+
import '../run-record-CPfd1ARZ.js';
|
|
5
5
|
import '../errors-CzMUYo7b.js';
|
|
6
6
|
import '../schema-m0gsnbt3.js';
|
|
7
7
|
import '../verdict-C9MlYujm.js';
|
package/dist/openapi.json
CHANGED
|
@@ -2,7 +2,7 @@
|
|
|
2
2
|
"openapi": "3.1.0",
|
|
3
3
|
"info": {
|
|
4
4
|
"title": "@tangle-network/agent-eval — wire protocol",
|
|
5
|
-
"version": "0.103.
|
|
5
|
+
"version": "0.103.1",
|
|
6
6
|
"description": "HTTP and stdio RPC interface to agent-eval. The TypeScript runtime is the source of truth; this spec is the contract that cross-language clients (Python, Rust, Go) generate from.\n\nWire-protocol version: 1.0.0. Bumps on breaking changes to request/response schemas.",
|
|
7
7
|
"contact": {
|
|
8
8
|
"name": "Tangle Network",
|
|
@@ -4,7 +4,7 @@ export { a as FailureCluster, F as FailureClusterReport, f as failureClusterView
|
|
|
4
4
|
import { a as TrajectoryStep } from '../trajectory-2TkpSEVh.js';
|
|
5
5
|
import { B as BaselineOptions, a as BaselineReport } from '../baseline-Bbid3WoO.js';
|
|
6
6
|
export { c as computeToolUseMetrics } from '../baseline-Bbid3WoO.js';
|
|
7
|
-
import { l as llmSpans } from '../query-
|
|
7
|
+
import { l as llmSpans } from '../query-0aTmbmQe.js';
|
|
8
8
|
|
|
9
9
|
/**
|
|
10
10
|
* BudgetBreachView — aggregates breach events across the corpus.
|
package/dist/pipelines/index.js
CHANGED
|
@@ -3,21 +3,21 @@ import {
|
|
|
3
3
|
classifyFailure,
|
|
4
4
|
compareToBaseline,
|
|
5
5
|
computeToolUseMetrics
|
|
6
|
-
} from "../chunk-
|
|
6
|
+
} from "../chunk-BXZVBX3D.js";
|
|
7
7
|
import {
|
|
8
8
|
buildTrajectory
|
|
9
9
|
} from "../chunk-RZTMDUO7.js";
|
|
10
10
|
import {
|
|
11
11
|
interRaterReliability,
|
|
12
12
|
pearsonR
|
|
13
|
-
} from "../chunk-
|
|
13
|
+
} from "../chunk-HYGRFL7C.js";
|
|
14
14
|
import {
|
|
15
15
|
aggregateLlm,
|
|
16
16
|
argHash,
|
|
17
17
|
llmSpans,
|
|
18
18
|
runFailureClass,
|
|
19
19
|
toolSpans
|
|
20
|
-
} from "../chunk-
|
|
20
|
+
} from "../chunk-JZXGWLK5.js";
|
|
21
21
|
import "../chunk-5BKGXME7.js";
|
|
22
22
|
import "../chunk-3BFEG2F6.js";
|
|
23
23
|
import "../chunk-PZ5AY32C.js";
|
|
@@ -1,6 +1,6 @@
|
|
|
1
|
-
import { A as AgentProfileCell, a as AgentProfileJson } from './run-record-
|
|
1
|
+
import { A as AgentProfileCell, a as AgentProfileJson } from './run-record-CPfd1ARZ.js';
|
|
2
2
|
import { V as ValidationError } from './errors-CzMUYo7b.js';
|
|
3
|
-
import { A as AnalystFinding, E as EvidenceRef } from './types-
|
|
3
|
+
import { A as AnalystFinding, E as EvidenceRef } from './types-DA9yj-Jd.js';
|
|
4
4
|
|
|
5
5
|
type PolicyEditSchemaVersion = 'policy-edit/v1';
|
|
6
6
|
declare const POLICY_EDIT_AXES: readonly ["carrier", "representation", "budget", "sampling", "output_contract", "tool_contract", "routing", "memory", "agent_profile", "deployment_target"];
|
|
@@ -1,7 +1,7 @@
|
|
|
1
1
|
import { A as AgentEvalError } from './errors-CzMUYo7b.js';
|
|
2
|
-
import { R as RunRecord } from './run-record-
|
|
3
|
-
import { C as ChatClient } from './types-
|
|
4
|
-
import { c as JudgeDimension, S as Scenario, a as JudgeConfig } from './types-
|
|
2
|
+
import { R as RunRecord } from './run-record-CPfd1ARZ.js';
|
|
3
|
+
import { C as ChatClient } from './types-DA9yj-Jd.js';
|
|
4
|
+
import { c as JudgeDimension, S as Scenario, a as JudgeConfig } from './types-Bihq6-a3.js';
|
|
5
5
|
import { TCloud } from '@tangle-network/tcloud';
|
|
6
6
|
import { R as RawProviderSink } from './raw-provider-sink-C46HDghv.js';
|
|
7
7
|
import { D as DefaultVerdict } from './verdict-C9MlYujm.js';
|
|
@@ -1,9 +1,9 @@
|
|
|
1
|
-
import { S as Scenario, g as Gate, G as GateResult, h as GateContext, C as CampaignResult, i as Mutator, f as SurfaceProposer, M as MutableSurface, j as GateDecision } from './types-
|
|
1
|
+
import { S as Scenario, g as Gate, G as GateResult, h as GateContext, C as CampaignResult, i as Mutator, f as SurfaceProposer, M as MutableSurface, j as GateDecision } from './types-Bihq6-a3.js';
|
|
2
2
|
import { R as RedTeamCase } from './red-team-BWdoyleI.js';
|
|
3
|
-
import { R as RunRecord } from './run-record-
|
|
3
|
+
import { R as RunRecord } from './run-record-CPfd1ARZ.js';
|
|
4
4
|
import { D as Direction } from './pareto-E-pembql.js';
|
|
5
|
-
import { a as PairedBootstrapResult } from './statistics-
|
|
6
|
-
import { R as RunCampaignOptions, C as CampaignStorage } from './gepa-
|
|
5
|
+
import { a as PairedBootstrapResult } from './statistics-D88peojY.js';
|
|
6
|
+
import { R as RunCampaignOptions, C as CampaignStorage } from './gepa-COlCAkHN.js';
|
|
7
7
|
import { HostedClient, TraceSpanEvent } from './hosted/index.js';
|
|
8
8
|
|
|
9
9
|
/**
|
|
@@ -72,9 +72,13 @@ interface DefaultProductionGateOptions {
|
|
|
72
72
|
* fires at the `gaming` severity. Default true. */
|
|
73
73
|
blockOnRewardHackingGaming?: boolean;
|
|
74
74
|
}
|
|
75
|
+
/**
|
|
76
|
+
* Opinionated production gate composing held-out significance, red-team, reward-hacking, and canary checks into a single `Gate.decide` decision.
|
|
77
|
+
*/
|
|
75
78
|
declare function defaultProductionGate<TArtifact, TScenario extends Scenario>(options: DefaultProductionGateOptions): Gate<TArtifact, TScenario>;
|
|
76
79
|
|
|
77
80
|
/**
|
|
81
|
+
* @module
|
|
78
82
|
* Thin Gate adapter — exposes delta-threshold-on-holdout as a composable
|
|
79
83
|
* `Gate`. Use when you want held-out as one of N composed gates instead of
|
|
80
84
|
* the full `defaultProductionGate` stack.
|
|
@@ -84,6 +88,9 @@ interface HeldOutGateOptions<TScenario extends Scenario = Scenario> {
|
|
|
84
88
|
scenarios: TScenario[];
|
|
85
89
|
deltaThreshold?: number;
|
|
86
90
|
}
|
|
91
|
+
/**
|
|
92
|
+
* Composable held-out delta gate: ships only when the candidate's mean composite on `scenarios` beats the baseline by at least `deltaThreshold`.
|
|
93
|
+
*/
|
|
87
94
|
declare function heldOutGate<TArtifact, TScenario extends Scenario>(options: HeldOutGateOptions<TScenario>): Gate<TArtifact, TScenario>;
|
|
88
95
|
|
|
89
96
|
/**
|
|
@@ -220,6 +227,9 @@ declare function paretoSignificanceGate<TArtifact = unknown, TScenario extends S
|
|
|
220
227
|
interface RunEvalOptions<TScenario extends Scenario, TArtifact> extends Omit<RunCampaignOptions<TScenario, TArtifact>, 'runDir'> {
|
|
221
228
|
runDir: string;
|
|
222
229
|
}
|
|
230
|
+
/**
|
|
231
|
+
* Simplest evaluation preset: run scenarios through dispatch, score with judges, and return a `CampaignResult` — no optimizer, no gate, no PR.
|
|
232
|
+
*/
|
|
223
233
|
declare function runEval<TScenario extends Scenario, TArtifact>(opts: RunEvalOptions<TScenario, TArtifact>): Promise<CampaignResult<TArtifact, TScenario>>;
|
|
224
234
|
|
|
225
235
|
/**
|
|
@@ -240,6 +250,9 @@ interface EvolutionaryProposerOptions<TFindings = unknown> {
|
|
|
240
250
|
/** External findings fed to the mutator each generation. Default: []. */
|
|
241
251
|
findings?: TFindings[];
|
|
242
252
|
}
|
|
253
|
+
/**
|
|
254
|
+
* Wrap a stateless `Mutator` (GEPA, AxGEPA, reflective-mutation) as a `SurfaceProposer` that mutates the current best surface into N candidates each generation.
|
|
255
|
+
*/
|
|
243
256
|
declare function evolutionaryProposer<TFindings = unknown>(opts: EvolutionaryProposerOptions<TFindings>): SurfaceProposer<TFindings>;
|
|
244
257
|
|
|
245
258
|
/**
|
|
@@ -396,6 +409,9 @@ declare function loopProvenanceSpans(record: LoopProvenanceRecord, opts?: {
|
|
|
396
409
|
}): TraceSpanEvent[];
|
|
397
410
|
/** Canonical durable paths under the run dir. */
|
|
398
411
|
declare function provenanceRecordPath(runDir: string): string;
|
|
412
|
+
/**
|
|
413
|
+
* Canonical path for the durable OTLP spans JSONL file under a loop run directory.
|
|
414
|
+
*/
|
|
399
415
|
declare function provenanceSpansPath(runDir: string): string;
|
|
400
416
|
interface EmitLoopProvenanceResult {
|
|
401
417
|
record: LoopProvenanceRecord;
|
|
@@ -13,6 +13,7 @@ import { T as TraceStore } from './store-BcFXE6LG.js';
|
|
|
13
13
|
declare function runsForScenario(store: TraceStore, scenarioId: string): Promise<Run[]>;
|
|
14
14
|
declare function llmSpans(store: TraceStore, runId?: string): Promise<LlmSpan[]>;
|
|
15
15
|
declare function toolSpans(store: TraceStore, runId?: string, toolName?: string): Promise<ToolSpan[]>;
|
|
16
|
+
/** Query judge-kind spans from the trace store, optionally scoped to a single run. */
|
|
16
17
|
declare function judgeSpans(store: TraceStore, runId?: string): Promise<JudgeSpan[]>;
|
|
17
18
|
/** Group spans by any key selector. */
|
|
18
19
|
declare function groupBy<T, K extends string | number>(items: T[], key: (t: T) => K): Map<K, T[]>;
|
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
import { D as DatasetSplit, c as DatasetManifest, a as DatasetScenario } from './dataset-BbGkaN2I.js';
|
|
2
|
-
import { m as GateDecision } from './summary-report-
|
|
3
|
-
import { R as RunRecord, b as RunSplitTag } from './run-record-
|
|
2
|
+
import { m as GateDecision } from './summary-report-Fc_YFJat.js';
|
|
3
|
+
import { R as RunRecord, b as RunSplitTag } from './run-record-CPfd1ARZ.js';
|
|
4
4
|
|
|
5
5
|
/**
|
|
6
6
|
* Release confidence gate.
|
|
@@ -216,6 +216,9 @@ interface JudgeReplayGateArgs<TOutput> {
|
|
|
216
216
|
/** Maximum concurrent judge calls. Default 4. */
|
|
217
217
|
judgeConcurrency?: number;
|
|
218
218
|
}
|
|
219
|
+
/**
|
|
220
|
+
* Confirm a candidate's win with a stronger judge: score baseline and candidate outputs independently, then bootstrap a CI to verify the lift generalises beyond the inner loop.
|
|
221
|
+
*/
|
|
219
222
|
declare function judgeReplayGate<TOutput>(args: JudgeReplayGateArgs<TOutput>): Promise<BootstrapResult & {
|
|
220
223
|
baselineSamples: number;
|
|
221
224
|
candidateSamples: number;
|
package/dist/reporting.d.ts
CHANGED
|
@@ -1,15 +1,15 @@
|
|
|
1
|
-
export { R as RubricOutcomePair, a as RubricPredictiveValidityInput, b as RubricPredictiveValidityReport, c as RubricRanking, r as rubricPredictiveValidity } from './rubric-predictive-validity-
|
|
2
|
-
export { B as BootstrapOptions, a as BootstrapResult, J as JudgeReplayGateArgs, R as ReleaseConfidenceAxis, b as ReleaseConfidenceAxisName, c as ReleaseConfidenceInput, d as ReleaseConfidenceIssue, e as ReleaseConfidenceMetrics, f as ReleaseConfidenceScorecard, g as ReleaseConfidenceStatus, h as ReleaseConfidenceThresholds, i as ReleaseTraceEvidence, j as RenderReleaseReportOptions, V as Verdict, k as assertReleaseConfidence, l as bootstrapCi, m as evaluateReleaseConfidence, n as judgeReplayGate, r as renderReleaseReport } from './release-report-
|
|
1
|
+
export { R as RubricOutcomePair, a as RubricPredictiveValidityInput, b as RubricPredictiveValidityReport, c as RubricRanking, r as rubricPredictiveValidity } from './rubric-predictive-validity-ZGIlJNce.js';
|
|
2
|
+
export { B as BootstrapOptions, a as BootstrapResult, J as JudgeReplayGateArgs, R as ReleaseConfidenceAxis, b as ReleaseConfidenceAxisName, c as ReleaseConfidenceInput, d as ReleaseConfidenceIssue, e as ReleaseConfidenceMetrics, f as ReleaseConfidenceScorecard, g as ReleaseConfidenceStatus, h as ReleaseConfidenceThresholds, i as ReleaseTraceEvidence, j as RenderReleaseReportOptions, V as Verdict, k as assertReleaseConfidence, l as bootstrapCi, m as evaluateReleaseConfidence, n as judgeReplayGate, r as renderReleaseReport } from './release-report-DkaCZ9k4.js';
|
|
3
3
|
export { I as InterimReleaseConfidence, a as InterimReleaseConfidenceInput, P as PairedEvalueOptions, b as PairedEvalueSequence, c as PairedEvalueStep, S as SequentialDecision, e as evaluateInterimReleaseConfidence, p as pairedEvalueSequence } from './sequential-5iSVfzl2.js';
|
|
4
|
-
export { P as PairedBootstrapOptions, a as PairedBootstrapResult, b as benjaminiHochberg, p as pairedBootstrap, w as wilcoxonSignedRank } from './statistics-
|
|
5
|
-
export { G as GainDistributionBin, a as GainDistributionFigureSpec, b as GainDistributionOptions, P as ParetoFigureSpec, c as ParetoPoint, R as RESEARCH_REPORT_HARD_PAIR_FLOOR, d as ResearchReport, e as ResearchReportCandidate, f as ResearchReportDecision, g as ResearchReportMethodology, h as ResearchReportOptions, i as ResearchReportRecommendation, S as SummaryTable, j as SummaryTableOptions, k as SummaryTableRow, l as gainHistogram, p as paretoChart, r as researchReport, s as summaryTable } from './summary-report-
|
|
6
|
-
import './run-record-
|
|
4
|
+
export { P as PairedBootstrapOptions, a as PairedBootstrapResult, b as benjaminiHochberg, p as pairedBootstrap, w as wilcoxonSignedRank } from './statistics-D88peojY.js';
|
|
5
|
+
export { G as GainDistributionBin, a as GainDistributionFigureSpec, b as GainDistributionOptions, P as ParetoFigureSpec, c as ParetoPoint, R as RESEARCH_REPORT_HARD_PAIR_FLOOR, d as ResearchReport, e as ResearchReportCandidate, f as ResearchReportDecision, g as ResearchReportMethodology, h as ResearchReportOptions, i as ResearchReportRecommendation, S as SummaryTable, j as SummaryTableOptions, k as SummaryTableRow, l as gainHistogram, p as paretoChart, r as researchReport, s as summaryTable } from './summary-report-Fc_YFJat.js';
|
|
6
|
+
import './run-record-CPfd1ARZ.js';
|
|
7
7
|
import '@tangle-network/agent-interface';
|
|
8
8
|
import './errors-CzMUYo7b.js';
|
|
9
9
|
import './schema-m0gsnbt3.js';
|
|
10
10
|
import './outcome-store-rnXLEqSn.js';
|
|
11
11
|
import './dataset-BbGkaN2I.js';
|
|
12
|
-
import './judge-calibration-
|
|
12
|
+
import './judge-calibration-7C-IDmKr.js';
|
|
13
13
|
import './types-C7DGg5ex.js';
|
|
14
14
|
import '@tangle-network/tcloud';
|
|
15
15
|
import './failure-cluster-DH9Flgcf.js';
|
package/dist/reporting.js
CHANGED
|
@@ -4,10 +4,10 @@ import {
|
|
|
4
4
|
evaluateReleaseConfidence,
|
|
5
5
|
judgeReplayGate,
|
|
6
6
|
renderReleaseReport
|
|
7
|
-
} from "./chunk-
|
|
7
|
+
} from "./chunk-UMEAR2FI.js";
|
|
8
8
|
import {
|
|
9
9
|
rubricPredictiveValidity
|
|
10
|
-
} from "./chunk-
|
|
10
|
+
} from "./chunk-QCBB6ZIU.js";
|
|
11
11
|
import {
|
|
12
12
|
evaluateInterimReleaseConfidence,
|
|
13
13
|
pairedEvalueSequence
|
|
@@ -18,12 +18,12 @@ import {
|
|
|
18
18
|
paretoChart,
|
|
19
19
|
researchReport,
|
|
20
20
|
summaryTable
|
|
21
|
-
} from "./chunk-
|
|
21
|
+
} from "./chunk-CLS3374R.js";
|
|
22
22
|
import {
|
|
23
23
|
benjaminiHochberg,
|
|
24
24
|
pairedBootstrap,
|
|
25
25
|
wilcoxonSignedRank
|
|
26
|
-
} from "./chunk-
|
|
26
|
+
} from "./chunk-HYGRFL7C.js";
|
|
27
27
|
import "./chunk-VSMTAMNK.js";
|
|
28
28
|
import "./chunk-3BFEG2F6.js";
|
|
29
29
|
import "./chunk-PZ5AY32C.js";
|
|
@@ -1,6 +1,6 @@
|
|
|
1
|
-
import { b as RunSplitTag, c as RunTokenUsage, d as RunJudgeMetadata, J as JudgeScoresRecord, A as AgentProfileCell, e as AgentProfileCellInput, R as RunRecord } from './run-record-
|
|
1
|
+
import { b as RunSplitTag, c as RunTokenUsage, d as RunJudgeMetadata, J as JudgeScoresRecord, A as AgentProfileCell, e as AgentProfileCellInput, R as RunRecord } from './run-record-CPfd1ARZ.js';
|
|
2
2
|
import { L as LlmClientOptions, a as LlmRouteRequirements } from './llm-client-Bj7g0rqu.js';
|
|
3
|
-
import { h as ResearchReportOptions, d as ResearchReport, m as GateDecision } from './summary-report-
|
|
3
|
+
import { h as ResearchReportOptions, d as ResearchReport, m as GateDecision } from './summary-report-Fc_YFJat.js';
|
|
4
4
|
import { T as TraceEmitter, R as RunCompleteHook } from './emitter-C2rqGH_l.js';
|
|
5
5
|
import { R as RunIntegrityExpectations, a as RunIntegrityReport } from './integrity-D2t12mMw.js';
|
|
6
6
|
import { R as RawProviderSink } from './raw-provider-sink-C46HDghv.js';
|
package/dist/rl.d.ts
CHANGED
|
@@ -1,23 +1,23 @@
|
|
|
1
|
-
import { R as RunRecord, b as RunSplitTag } from './run-record-
|
|
1
|
+
import { R as RunRecord, b as RunSplitTag } from './run-record-CPfd1ARZ.js';
|
|
2
2
|
export { A as AdversarialMutation } from './adversarial-B7loGVVX.js';
|
|
3
|
-
import { P as PreferenceExtractionReport, D as DpoExportRow, G as GrpoExportRow, S as SftExportRow, E as ExtractPreferencesOptions, a as DpoLookups, b as GrpoLookups, c as SftLookups } from './corpus-
|
|
4
|
-
export { d as CorpusAppendResult, C as CorpusRecord, e as DatasetFormat, f as ExtractStepRewardsOptions, H as HarvestOptions, g as PreferenceStrategy, h as PreferenceTriple, i as PrmExportRow, j as PrmLookups, k as PrmTrainingTriple, R as RewardKind, l as RewardStats, m as RlDatasetBundle, n as RlDatasetConfig, o as RlDatasetManifest, p as RlDatasetStats, q as RunwiseStepSummary, r as StepReward, s as StepRewardJsonlRow, t as StepScorer, u as appendToCorpus, v as buildDatasetFromCorpus, w as buildRlDataset, x as datasheetToMarkdown, y as extractPreferences, z as extractStepRewards, A as prmTrainingPairs, B as readCorpus, F as runwiseStepRewardSummary, I as stepRewardsToJsonl, J as toAnthropicFormat, K as toDpoJsonl, L as toDpoRows, M as toGrpoJsonl, N as toGrpoRows, O as toPrmJsonl, Q as toPrmRows, T as toSftJsonl, U as toSftRows, V as toTRLFormat } from './corpus-
|
|
3
|
+
import { P as PreferenceExtractionReport, D as DpoExportRow, G as GrpoExportRow, S as SftExportRow, E as ExtractPreferencesOptions, a as DpoLookups, b as GrpoLookups, c as SftLookups } from './corpus-CysLCiK4.js';
|
|
4
|
+
export { d as CorpusAppendResult, C as CorpusRecord, e as DatasetFormat, f as ExtractStepRewardsOptions, H as HarvestOptions, g as PreferenceStrategy, h as PreferenceTriple, i as PrmExportRow, j as PrmLookups, k as PrmTrainingTriple, R as RewardKind, l as RewardStats, m as RlDatasetBundle, n as RlDatasetConfig, o as RlDatasetManifest, p as RlDatasetStats, q as RunwiseStepSummary, r as StepReward, s as StepRewardJsonlRow, t as StepScorer, u as appendToCorpus, v as buildDatasetFromCorpus, w as buildRlDataset, x as datasheetToMarkdown, y as extractPreferences, z as extractStepRewards, A as prmTrainingPairs, B as readCorpus, F as runwiseStepRewardSummary, I as stepRewardsToJsonl, J as toAnthropicFormat, K as toDpoJsonl, L as toDpoRows, M as toGrpoJsonl, N as toGrpoRows, O as toPrmJsonl, Q as toPrmRows, T as toSftJsonl, U as toSftRows, V as toTRLFormat } from './corpus-CysLCiK4.js';
|
|
5
5
|
export { O as OffPolicyEstimate, a as OffPolicyOptions, b as OffPolicyTrajectory, d as doublyRobust, i as inverseProbabilityWeighting, o as offPolicyEstimateAll, s as selfNormalizedImportanceWeighting } from './off-policy-DiwuKKg7.js';
|
|
6
6
|
import { b as OutcomeStore } from './outcome-store-rnXLEqSn.js';
|
|
7
7
|
export { D as DeploymentOutcome, F as FileSystemOutcomeStore, a as FileSystemOutcomeStoreOptions, I as InMemoryOutcomeStore } from './outcome-store-rnXLEqSn.js';
|
|
8
|
-
import { b as RubricPredictiveValidityReport } from './rubric-predictive-validity-
|
|
9
|
-
import { R as Researcher, F as FailureMode, S as SteeringChange, E as ExperimentPlan, a as ExperimentResult, b as EvalCampaignResult, c as EvalCampaignOptions } from './researcher-
|
|
10
|
-
export { r as runEvalCampaign } from './researcher-
|
|
11
|
-
import { a as VerificationReport } from './multi-layer-verifier-
|
|
8
|
+
import { b as RubricPredictiveValidityReport } from './rubric-predictive-validity-ZGIlJNce.js';
|
|
9
|
+
import { R as Researcher, F as FailureMode, S as SteeringChange, E as ExperimentPlan, a as ExperimentResult, b as EvalCampaignResult, c as EvalCampaignOptions } from './researcher-SDAezfML.js';
|
|
10
|
+
export { r as runEvalCampaign } from './researcher-SDAezfML.js';
|
|
11
|
+
import { a as VerificationReport } from './multi-layer-verifier-CI4jdX-q.js';
|
|
12
12
|
import { I as InterimReleaseConfidence } from './sequential-5iSVfzl2.js';
|
|
13
|
-
import { C as CampaignResult } from './types-
|
|
13
|
+
import { C as CampaignResult } from './types-Bihq6-a3.js';
|
|
14
14
|
import '@tangle-network/agent-interface';
|
|
15
15
|
import './errors-CzMUYo7b.js';
|
|
16
16
|
import './schema-m0gsnbt3.js';
|
|
17
17
|
import './store-BcFXE6LG.js';
|
|
18
18
|
import './llm-client-Bj7g0rqu.js';
|
|
19
19
|
import './raw-provider-sink-C46HDghv.js';
|
|
20
|
-
import './summary-report-
|
|
20
|
+
import './summary-report-Fc_YFJat.js';
|
|
21
21
|
import './failure-cluster-DH9Flgcf.js';
|
|
22
22
|
import './emitter-C2rqGH_l.js';
|
|
23
23
|
import './integrity-D2t12mMw.js';
|
package/dist/rl.js
CHANGED
|
@@ -9,26 +9,26 @@ import {
|
|
|
9
9
|
extractVerifiableReward,
|
|
10
10
|
extractVerifiableRewardsFromRecords,
|
|
11
11
|
filterDeterministicallyRewarded
|
|
12
|
-
} from "./chunk-
|
|
12
|
+
} from "./chunk-LOZOZYHU.js";
|
|
13
13
|
import {
|
|
14
14
|
FileSystemOutcomeStore,
|
|
15
15
|
InMemoryOutcomeStore
|
|
16
16
|
} from "./chunk-3RF76KTD.js";
|
|
17
17
|
import {
|
|
18
18
|
runEvalCampaign
|
|
19
|
-
} from "./chunk-
|
|
19
|
+
} from "./chunk-6YVAN6R3.js";
|
|
20
20
|
import "./chunk-CWNP4DV4.js";
|
|
21
21
|
import {
|
|
22
22
|
rubricPredictiveValidity
|
|
23
|
-
} from "./chunk-
|
|
23
|
+
} from "./chunk-QCBB6ZIU.js";
|
|
24
24
|
import {
|
|
25
25
|
evaluateInterimReleaseConfidence
|
|
26
26
|
} from "./chunk-MAZ26DC7.js";
|
|
27
|
-
import "./chunk-
|
|
27
|
+
import "./chunk-CLS3374R.js";
|
|
28
28
|
import {
|
|
29
29
|
benjaminiHochberg,
|
|
30
30
|
wilcoxonSignedRank
|
|
31
|
-
} from "./chunk-
|
|
31
|
+
} from "./chunk-HYGRFL7C.js";
|
|
32
32
|
import {
|
|
33
33
|
observationsFromRunRecords,
|
|
34
34
|
thompsonCurriculum,
|
|
@@ -36,9 +36,9 @@ import {
|
|
|
36
36
|
} from "./chunk-VZSRQ272.js";
|
|
37
37
|
import "./chunk-SBCB6VZY.js";
|
|
38
38
|
import "./chunk-PC4UYEBM.js";
|
|
39
|
-
import "./chunk-
|
|
39
|
+
import "./chunk-J5MUFXEY.js";
|
|
40
40
|
import "./chunk-TVVP3ZZQ.js";
|
|
41
|
-
import "./chunk-
|
|
41
|
+
import "./chunk-AGYDMORK.js";
|
|
42
42
|
import "./chunk-VSMTAMNK.js";
|
|
43
43
|
import {
|
|
44
44
|
ValidationError
|
|
@@ -1,12 +1,12 @@
|
|
|
1
1
|
import {
|
|
2
2
|
planCampaignRun,
|
|
3
3
|
runCampaign
|
|
4
|
-
} from "./chunk-
|
|
5
|
-
import "./chunk-
|
|
4
|
+
} from "./chunk-VNM52AGA.js";
|
|
5
|
+
import "./chunk-HYGRFL7C.js";
|
|
6
6
|
import "./chunk-3BFEG2F6.js";
|
|
7
7
|
import "./chunk-PZ5AY32C.js";
|
|
8
8
|
export {
|
|
9
9
|
planCampaignRun,
|
|
10
10
|
runCampaign
|
|
11
11
|
};
|
|
12
|
-
//# sourceMappingURL=run-campaign-
|
|
12
|
+
//# sourceMappingURL=run-campaign-DAHKO5CT.js.map
|
|
@@ -49,6 +49,9 @@ declare class AgentProfileCellValidationError extends ValidationError {
|
|
|
49
49
|
}
|
|
50
50
|
declare function buildAgentProfileCell(input: AgentProfileCellInput): Promise<AgentProfileCell>;
|
|
51
51
|
declare function agentProfileCellHashMaterial(cell: AgentProfileCell): Omit<AgentProfileCell, 'cellId'>;
|
|
52
|
+
/**
|
|
53
|
+
* Verify an `AgentProfileCell`'s `cellId` matches the sha256 of its hash-material fields, confirming the record has not been tampered with.
|
|
54
|
+
*/
|
|
52
55
|
declare function verifyAgentProfileCell(cell: AgentProfileCell): Promise<boolean>;
|
|
53
56
|
declare function validateAgentProfileCell(input: unknown): AgentProfileCell;
|
|
54
57
|
declare function requireAgentProfileCell(record: {
|
|
@@ -1,10 +1,10 @@
|
|
|
1
1
|
import { AxAIService } from '@ax-llm/ax';
|
|
2
2
|
import { z } from 'zod';
|
|
3
|
-
import { A as AnalystFinding, a as Analyst, b as AnalystContext } from './types-
|
|
4
|
-
import { T as TraceAnalystKindSpec } from './kind-factory-
|
|
3
|
+
import { A as AnalystFinding, a as Analyst, b as AnalystContext } from './types-DA9yj-Jd.js';
|
|
4
|
+
import { T as TraceAnalystKindSpec } from './kind-factory-BLHxwX71.js';
|
|
5
5
|
import { a as TraceAnalystSpan } from './store-C1YxJDEK.js';
|
|
6
6
|
import { L as LlmClientOptions } from './llm-client-Bj7g0rqu.js';
|
|
7
|
-
import { S as Severity } from './multi-layer-verifier-
|
|
7
|
+
import { S as Severity } from './multi-layer-verifier-CI4jdX-q.js';
|
|
8
8
|
|
|
9
9
|
interface CreateAnalystAiConfig {
|
|
10
10
|
/** OpenAI-compatible API key forwarded as `Authorization: Bearer`.
|
|
@@ -1,4 +1,4 @@
|
|
|
1
|
-
import { d as ContinuousAgreementOptions, C as ContinuousAgreement } from './judge-calibration-
|
|
1
|
+
import { d as ContinuousAgreementOptions, C as ContinuousAgreement } from './judge-calibration-7C-IDmKr.js';
|
|
2
2
|
import { J as JudgeScore } from './types-C7DGg5ex.js';
|
|
3
3
|
|
|
4
4
|
/** Identity: dimensions already follow "higher = better" by prompt convention
|
package/dist/traces.d.ts
CHANGED
|
@@ -6,7 +6,7 @@ export { S as SpanHandle, T as TraceEmitter, b as TraceEmitterOptions, l as llmS
|
|
|
6
6
|
export { b as RunIntegrityError, R as RunIntegrityExpectations, c as RunIntegrityIssue, d as RunIntegrityIssueCode, a as RunIntegrityReport, e as assertRunCaptured, t as throwIfRunIncomplete } from './integrity-D2t12mMw.js';
|
|
7
7
|
import { T as TraceStore } from './store-BcFXE6LG.js';
|
|
8
8
|
export { E as EventFilter, F as FileSystemTraceStore, a as FileSystemTraceStoreOptions, I as InMemoryTraceStore, R as RunFilter, S as SpanFilter } from './store-BcFXE6LG.js';
|
|
9
|
-
export { a as aggregateLlm, b as argHash, g as groupBy, j as judgeSpans, l as llmSpans, r as runFailureClass, c as runsForScenario, t as toolSpans } from './query-
|
|
9
|
+
export { a as aggregateLlm, b as argHash, g as groupBy, j as judgeSpans, l as llmSpans, r as runFailureClass, c as runsForScenario, t as toolSpans } from './query-0aTmbmQe.js';
|
|
10
10
|
export { D as DEFAULT_REDACTION_RULES, b as REDACTION_VERSION, a as RedactionReport, R as RedactionRule, r as redactString, c as redactValue } from './redact-B40YG2M_.js';
|
|
11
11
|
import { R as Run } from './schema-m0gsnbt3.js';
|
|
12
12
|
export { A as Artifact, B as BudgetLedgerEntry, h as BudgetSpec, E as EventKind, i as FAILURE_CLASSES, F as FailureClass, G as GenericSpan, J as JudgeSpan, L as LlmSpan, M as Message, d as RetrievalSpan, g as RunLayer, b as RunOutcome, f as RunStatus, e as SandboxSpan, S as Span, j as SpanBase, c as SpanKind, k as SpanStatus, l as TRACE_SCHEMA_VERSION, T as ToolSpan, a as TraceEvent, m as isJudgeSpan, n as isLlmSpan, o as isRetrievalSpan, p as isSandboxSpan, q as isToolSpan } from './schema-m0gsnbt3.js';
|
|
@@ -14,7 +14,7 @@ import { A as AnalyzeTracesOptions, b as AnalyzeTracesResult } from './analyst-C
|
|
|
14
14
|
export { a as AnalyzeTracesInput, c as AnalyzeTracesTurnSnapshot, d as analyzeTraces } from './analyst-C8HHvfJp.js';
|
|
15
15
|
import { h as TraceAnalystSpanKind, i as TraceAnalystSpanStatus, T as TraceAnalysisStore, g as TraceAnalystFilters, b as DatasetOverview, Q as QueryTracesPage, l as ViewTraceResult, V as ViewSpansResult, c as SearchTraceResult, S as SearchSpanResult } from './store-C1YxJDEK.js';
|
|
16
16
|
export { D as DEFAULT_TRACE_ANALYST_BUDGETS, E as ErrorCluster, d as SpanMatchRecord, e as TRACE_ANALYST_TRUNCATION_MARKER_PREFIX, f as TraceAnalystByteBudgets, a as TraceAnalystSpan, j as TraceAnalystTraceSummary, k as ViewTraceOversized } from './store-C1YxJDEK.js';
|
|
17
|
-
import { b as RunSplitTag, c as RunTokenUsage, R as RunRecord } from './run-record-
|
|
17
|
+
import { b as RunSplitTag, c as RunTokenUsage, R as RunRecord } from './run-record-CPfd1ARZ.js';
|
|
18
18
|
import { AxFunction } from '@ax-llm/ax';
|
|
19
19
|
import '@tangle-network/agent-interface';
|
|
20
20
|
|
package/dist/traces.js
CHANGED
|
@@ -28,7 +28,7 @@ import {
|
|
|
28
28
|
scoreTraceInsightReadiness,
|
|
29
29
|
tokenizeDomainWords,
|
|
30
30
|
traceAnalystOnRunComplete
|
|
31
|
-
} from "./chunk-
|
|
31
|
+
} from "./chunk-3722PKPB.js";
|
|
32
32
|
import {
|
|
33
33
|
TRACE_ANALYST_ACTOR_DESCRIPTION,
|
|
34
34
|
TRACE_ANALYST_ACTOR_DESCRIPTION_VERSION,
|
|
@@ -50,7 +50,7 @@ import {
|
|
|
50
50
|
runFailureClass,
|
|
51
51
|
runsForScenario,
|
|
52
52
|
toolSpans
|
|
53
|
-
} from "./chunk-
|
|
53
|
+
} from "./chunk-JZXGWLK5.js";
|
|
54
54
|
import {
|
|
55
55
|
FAILURE_CLASSES,
|
|
56
56
|
TRACE_SCHEMA_VERSION,
|
|
@@ -106,12 +106,12 @@ import {
|
|
|
106
106
|
defaultProviderRedactor,
|
|
107
107
|
providerFromBaseUrl
|
|
108
108
|
} from "./chunk-PC4UYEBM.js";
|
|
109
|
-
import "./chunk-
|
|
109
|
+
import "./chunk-J5MUFXEY.js";
|
|
110
110
|
import {
|
|
111
111
|
TraceEmitter,
|
|
112
112
|
llmSpanFromProvider
|
|
113
113
|
} from "./chunk-TVVP3ZZQ.js";
|
|
114
|
-
import "./chunk-
|
|
114
|
+
import "./chunk-AGYDMORK.js";
|
|
115
115
|
import "./chunk-VSMTAMNK.js";
|
|
116
116
|
import "./chunk-3BFEG2F6.js";
|
|
117
117
|
import "./chunk-PZ5AY32C.js";
|
|
@@ -1,4 +1,4 @@
|
|
|
1
|
-
import { R as RunRecord } from './run-record-
|
|
1
|
+
import { R as RunRecord } from './run-record-CPfd1ARZ.js';
|
|
2
2
|
import { T as TraceAnalysisStore } from './store-C1YxJDEK.js';
|
|
3
3
|
import { a as JudgeInput } from './types-C7DGg5ex.js';
|
|
4
4
|
import { b as LlmCallRequest, c as LlmCallResult } from './llm-client-Bj7g0rqu.js';
|
package/dist/workflow/index.d.ts
CHANGED
|
@@ -1,13 +1,13 @@
|
|
|
1
1
|
import { W as WorkflowTopology } from '../harness-optimizer-mOl9XX_O.js';
|
|
2
|
-
import { b as RunSplitTag, c as RunTokenUsage, R as RunRecord } from '../run-record-
|
|
3
|
-
import { A as AnalystFinding, h as AnalystSeverity, E as EvidenceRef } from '../types-
|
|
4
|
-
import { F as FailureClusterInsight } from '../insight-report-
|
|
5
|
-
import { a as VerificationReport, L as LayerResult } from '../multi-layer-verifier-
|
|
2
|
+
import { b as RunSplitTag, c as RunTokenUsage, R as RunRecord } from '../run-record-CPfd1ARZ.js';
|
|
3
|
+
import { A as AnalystFinding, h as AnalystSeverity, E as EvidenceRef } from '../types-DA9yj-Jd.js';
|
|
4
|
+
import { F as FailureClusterInsight } from '../insight-report-DumCfEur.js';
|
|
5
|
+
import { a as VerificationReport, L as LayerResult } from '../multi-layer-verifier-CI4jdX-q.js';
|
|
6
6
|
import { F as FailureClusterReport } from '../failure-cluster-DH9Flgcf.js';
|
|
7
7
|
import { R as RedactionRule, a as RedactionReport } from '../redact-B40YG2M_.js';
|
|
8
8
|
import { D as DatasetSplit } from '../dataset-BbGkaN2I.js';
|
|
9
9
|
import { a as FeedbackTrajectory } from '../feedback-trajectory-BxY0cKfs.js';
|
|
10
|
-
import { a as PairedBootstrapResult } from '../statistics-
|
|
10
|
+
import { a as PairedBootstrapResult } from '../statistics-D88peojY.js';
|
|
11
11
|
import '../pareto-E-pembql.js';
|
|
12
12
|
import '../run-critic-CmMf05uV.js';
|
|
13
13
|
import '../schema-m0gsnbt3.js';
|
|
@@ -19,8 +19,8 @@ import '../types-C7DGg5ex.js';
|
|
|
19
19
|
import '@tangle-network/tcloud';
|
|
20
20
|
import '../llm-client-Bj7g0rqu.js';
|
|
21
21
|
import '../raw-provider-sink-C46HDghv.js';
|
|
22
|
-
import '../summary-report-
|
|
23
|
-
import '../judge-calibration-
|
|
22
|
+
import '../summary-report-Fc_YFJat.js';
|
|
23
|
+
import '../judge-calibration-7C-IDmKr.js';
|
|
24
24
|
import '../verdict-C9MlYujm.js';
|
|
25
25
|
import '../control-runtime-Acf9CGhw.js';
|
|
26
26
|
import '../emitter-C2rqGH_l.js';
|
package/dist/workflow/index.js
CHANGED
|
@@ -1,14 +1,14 @@
|
|
|
1
1
|
import {
|
|
2
2
|
pairedBootstrap
|
|
3
|
-
} from "../chunk-
|
|
3
|
+
} from "../chunk-HYGRFL7C.js";
|
|
4
4
|
import {
|
|
5
5
|
DEFAULT_REDACTION_RULES,
|
|
6
6
|
redactString
|
|
7
7
|
} from "../chunk-GGE4NNQT.js";
|
|
8
8
|
import {
|
|
9
9
|
validateRunRecord
|
|
10
|
-
} from "../chunk-
|
|
11
|
-
import "../chunk-
|
|
10
|
+
} from "../chunk-J5MUFXEY.js";
|
|
11
|
+
import "../chunk-AGYDMORK.js";
|
|
12
12
|
import "../chunk-VSMTAMNK.js";
|
|
13
13
|
import {
|
|
14
14
|
ValidationError
|
package/package.json
CHANGED
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
{
|
|
2
2
|
"name": "@tangle-network/agent-eval",
|
|
3
|
-
"version": "0.103.
|
|
3
|
+
"version": "0.103.1",
|
|
4
4
|
"description": "Evaluate and improve AI agents from runs, traces, judges, and feedback. Compare candidates, cluster failures, measure lift, and gate releases.",
|
|
5
5
|
"homepage": "https://github.com/tangle-network/agent-eval#readme",
|
|
6
6
|
"repository": {
|