@tangle-network/agent-eval 0.102.1 → 0.103.1
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/dist/adapters/http.d.ts +2 -2
- package/dist/adapters/langchain.d.ts +2 -2
- package/dist/adapters/otel.d.ts +5 -5
- package/dist/analyst/index.d.ts +9 -9
- package/dist/analyst/index.js +3 -3
- package/dist/{analyze-runs-Cd-A_K4l.d.ts → analyze-runs-rF2eGxZ_.d.ts} +3 -3
- package/dist/belief-state/index.d.ts +3 -3
- package/dist/belief-state/index.js +1 -1
- package/dist/benchmarks/index.d.ts +2 -2
- package/dist/builder-eval/index.js +2 -2
- package/dist/campaign/index.d.ts +43 -16
- package/dist/campaign/index.js +9 -9
- package/dist/campaign/index.js.map +1 -1
- package/dist/{chunk-Z3FLN24V.js → chunk-2B4HPFXN.js} +104 -43
- package/dist/chunk-2B4HPFXN.js.map +1 -0
- package/dist/{chunk-7RBJANJD.js → chunk-3722PKPB.js} +2 -2
- package/dist/{chunk-YLKDN7JV.js → chunk-66T25VWE.js} +5 -5
- package/dist/chunk-66T25VWE.js.map +1 -0
- package/dist/{chunk-6FIAJHCU.js → chunk-6YVAN6R3.js} +4 -4
- package/dist/{chunk-ABOIVNXL.js → chunk-AGYDMORK.js} +1 -1
- package/dist/chunk-AGYDMORK.js.map +1 -0
- package/dist/{chunk-B2TMQM62.js → chunk-BHCFJGL4.js} +2 -2
- package/dist/{chunk-IH7LYRHL.js → chunk-BXZVBX3D.js} +3 -3
- package/dist/{chunk-6GT4NI4V.js → chunk-CLS3374R.js} +2 -2
- package/dist/{chunk-HRGUJTER.js → chunk-HYGRFL7C.js} +1 -1
- package/dist/{chunk-HRGUJTER.js.map → chunk-HYGRFL7C.js.map} +1 -1
- package/dist/{chunk-2NSLDY4B.js → chunk-J5MUFXEY.js} +2 -2
- package/dist/{chunk-UFSG7ACU.js → chunk-JZXGWLK5.js} +1 -1
- package/dist/chunk-JZXGWLK5.js.map +1 -0
- package/dist/{chunk-IN3SHQML.js → chunk-LMSQ6EFA.js} +2 -2
- package/dist/{chunk-AIGWQEME.js → chunk-LOZOZYHU.js} +2 -2
- package/dist/{chunk-NF7OZ4J7.js → chunk-QCBB6ZIU.js} +2 -2
- package/dist/chunk-RQNOLV3I.js +855 -0
- package/dist/chunk-RQNOLV3I.js.map +1 -0
- package/dist/{chunk-JU6ZX3CX.js → chunk-SMCACT4Z.js} +2 -2
- package/dist/{chunk-6Q2DYRWV.js → chunk-TCPP4Z5Q.js} +5 -5
- package/dist/{chunk-VDGPPGE3.js → chunk-UMEAR2FI.js} +2 -2
- package/dist/{chunk-VDGPPGE3.js.map → chunk-UMEAR2FI.js.map} +1 -1
- package/dist/{chunk-QIT2XZ4E.js → chunk-VNM52AGA.js} +3 -3
- package/dist/chunk-VNM52AGA.js.map +1 -0
- package/dist/{chunk-CMJSTXUR.js → chunk-YFYIOSNV.js} +107 -20
- package/dist/chunk-YFYIOSNV.js.map +1 -0
- package/dist/{code-agent-session-Ce-9u7YM.d.ts → code-agent-session-Cen-qD2y.d.ts} +1 -1
- package/dist/contract/index.d.ts +18 -18
- package/dist/contract/index.js +10 -10
- package/dist/{control-C8RmK9H4.d.ts → control-KofK3gfG.d.ts} +1 -1
- package/dist/control.d.ts +2 -2
- package/dist/control.js +3 -3
- package/dist/{corpus-CiSzzLa5.d.ts → corpus-CysLCiK4.d.ts} +1 -1
- package/dist/{default-registry-ZhqsTr4K.d.ts → default-registry-Ez4cxuZ4.d.ts} +2 -2
- package/dist/diagnose.d.ts +3 -3
- package/dist/diagnose.js +3 -3
- package/dist/{gepa-DeyPTlvx.d.ts → gepa-COlCAkHN.d.ts} +22 -1
- package/dist/governance/index.d.ts +1 -1
- package/dist/hosted/index.d.ts +5 -5
- package/dist/{index-B-bFgiAF.d.ts → index-unCSYRJJ.d.ts} +1 -1
- package/dist/index.d.ts +36 -28
- package/dist/index.js +32 -18
- package/dist/index.js.map +1 -1
- package/dist/{insight-report-k0sRTzKg.d.ts → insight-report-DumCfEur.d.ts} +2 -2
- package/dist/{judge-calibration-0p2QcWNE.d.ts → judge-calibration-7C-IDmKr.d.ts} +3 -0
- package/dist/{kind-factory-D0nk7AKV.d.ts → kind-factory-BLHxwX71.d.ts} +1 -1
- package/dist/meta-eval/index.d.ts +4 -4
- package/dist/meta-eval/index.js +3 -3
- package/dist/{multi-layer-verifier-DUZXrPDA.d.ts → multi-layer-verifier-CI4jdX-q.d.ts} +3 -0
- package/dist/multishot/index.d.ts +2 -2
- package/dist/openapi.json +1 -1
- package/dist/pipelines/index.d.ts +1 -1
- package/dist/pipelines/index.js +3 -3
- package/dist/{policy-edit-BDQzzsBU.d.ts → policy-edit-nUhuFtLF.d.ts} +2 -2
- package/dist/{pre-registration-Dzg61IQA.d.ts → pre-registration-CUOSGAZK.d.ts} +45 -12
- package/dist/product-benchmark/index.d.ts +104 -1
- package/dist/product-benchmark/index.js +15 -1
- package/dist/{provenance-BhJm32vN.d.ts → provenance-B2VsP0jP.d.ts} +20 -4
- package/dist/{query-B7GGjRox.d.ts → query-0aTmbmQe.d.ts} +1 -0
- package/dist/{release-report-BQ1Ziyu-.d.ts → release-report-DkaCZ9k4.d.ts} +5 -2
- package/dist/reporting.d.ts +6 -6
- package/dist/reporting.js +4 -4
- package/dist/{researcher-B_ODTAJs.d.ts → researcher-SDAezfML.d.ts} +2 -2
- package/dist/rl.d.ts +9 -9
- package/dist/rl.js +7 -7
- package/dist/{rubric-predictive-validity-0MdjTt8R.d.ts → rubric-predictive-validity-ZGIlJNce.d.ts} +1 -1
- package/dist/{run-campaign-3NWW5PLF.js → run-campaign-DAHKO5CT.js} +3 -3
- package/dist/{run-record-MRdJ-Kq2.d.ts → run-record-CPfd1ARZ.d.ts} +3 -0
- package/dist/{runtime-trajectory-8w0_jmtR.d.ts → runtime-trajectory-DZ8ei-Jo.d.ts} +1 -1
- package/dist/{semantic-concept-judge-D-IlH5v1.d.ts → semantic-concept-judge-C6mDEBIo.d.ts} +3 -3
- package/dist/{statistics-xP-cWc5k.d.ts → statistics-D88peojY.d.ts} +1 -1
- package/dist/{summary-report-C0nnxOD8.d.ts → summary-report-Fc_YFJat.d.ts} +1 -1
- package/dist/traces.d.ts +2 -2
- package/dist/traces.js +4 -4
- package/dist/{types-DFI_Z-ZL.d.ts → types-Bihq6-a3.d.ts} +1 -1
- package/dist/{types-Dz9cKF0g.d.ts → types-DA9yj-Jd.d.ts} +1 -1
- package/dist/workflow/index.d.ts +7 -7
- package/dist/workflow/index.js +3 -3
- package/package.json +1 -1
- package/dist/chunk-63MBSQTX.js +0 -350
- package/dist/chunk-63MBSQTX.js.map +0 -1
- package/dist/chunk-ABOIVNXL.js.map +0 -1
- package/dist/chunk-CMJSTXUR.js.map +0 -1
- package/dist/chunk-QIT2XZ4E.js.map +0 -1
- package/dist/chunk-UFSG7ACU.js.map +0 -1
- package/dist/chunk-YLKDN7JV.js.map +0 -1
- package/dist/chunk-Z3FLN24V.js.map +0 -1
- /package/dist/{chunk-7RBJANJD.js.map → chunk-3722PKPB.js.map} +0 -0
- /package/dist/{chunk-6FIAJHCU.js.map → chunk-6YVAN6R3.js.map} +0 -0
- /package/dist/{chunk-B2TMQM62.js.map → chunk-BHCFJGL4.js.map} +0 -0
- /package/dist/{chunk-IH7LYRHL.js.map → chunk-BXZVBX3D.js.map} +0 -0
- /package/dist/{chunk-6GT4NI4V.js.map → chunk-CLS3374R.js.map} +0 -0
- /package/dist/{chunk-2NSLDY4B.js.map → chunk-J5MUFXEY.js.map} +0 -0
- /package/dist/{chunk-IN3SHQML.js.map → chunk-LMSQ6EFA.js.map} +0 -0
- /package/dist/{chunk-AIGWQEME.js.map → chunk-LOZOZYHU.js.map} +0 -0
- /package/dist/{chunk-NF7OZ4J7.js.map → chunk-QCBB6ZIU.js.map} +0 -0
- /package/dist/{chunk-JU6ZX3CX.js.map → chunk-SMCACT4Z.js.map} +0 -0
- /package/dist/{chunk-6Q2DYRWV.js.map → chunk-TCPP4Z5Q.js.map} +0 -0
- /package/dist/{run-campaign-3NWW5PLF.js.map → run-campaign-DAHKO5CT.js.map} +0 -0
|
@@ -1,5 +1,5 @@
|
|
|
1
|
-
import { G as GainDistributionBin, P as ParetoFigureSpec } from './summary-report-
|
|
2
|
-
import { C as ContinuousAgreement } from './judge-calibration-
|
|
1
|
+
import { G as GainDistributionBin, P as ParetoFigureSpec } from './summary-report-Fc_YFJat.js';
|
|
2
|
+
import { C as ContinuousAgreement } from './judge-calibration-7C-IDmKr.js';
|
|
3
3
|
|
|
4
4
|
/**
|
|
5
5
|
* # InsightReport — the rigorous decision packet for any set of agent runs.
|
|
@@ -45,6 +45,9 @@ interface CalibrationResult {
|
|
|
45
45
|
delta: number;
|
|
46
46
|
}>;
|
|
47
47
|
}
|
|
48
|
+
/**
|
|
49
|
+
* Measure judge quality against human gold labels: computes Cohen's κ, Pearson correlation, and MAE over matched item ids.
|
|
50
|
+
*/
|
|
48
51
|
declare function calibrateJudge(golden: GoldenItem[], candidate: CandidateScore[]): CalibrationResult;
|
|
49
52
|
interface PositionalBiasResult {
|
|
50
53
|
/**
|
|
@@ -1,7 +1,7 @@
|
|
|
1
1
|
import { AxAIService, AxFunction } from '@ax-llm/ax';
|
|
2
2
|
import { T as TraceAnalysisStore } from './store-C1YxJDEK.js';
|
|
3
3
|
import { z } from 'zod';
|
|
4
|
-
import { g as AnalystCost, b as AnalystContext, a as Analyst } from './types-
|
|
4
|
+
import { g as AnalystCost, b as AnalystContext, a as Analyst } from './types-DA9yj-Jd.js';
|
|
5
5
|
|
|
6
6
|
/**
|
|
7
7
|
* Typed Ax output for analyst findings.
|
|
@@ -1,12 +1,12 @@
|
|
|
1
1
|
export { C as CalibrationBin, a as CalibrationOptions, b as CalibrationPair, c as CalibrationReport, d as CorrelationResult, e as CorrelationStudyOptions, f as CorrelationStudyResult, E as EvalMetricSpec, O as OutcomePair, g as calibrationCurve, h as calibrationFromPairs, i as correlationStudy } from '../calibration-BPmzuVPk.js';
|
|
2
2
|
export { D as DeploymentOutcome, F as FileSystemOutcomeStore, a as FileSystemOutcomeStoreOptions, I as InMemoryOutcomeStore, O as OutcomeFilter, b as OutcomeStore } from '../outcome-store-rnXLEqSn.js';
|
|
3
|
-
export { R as RubricOutcomePair, a as RubricPredictiveValidityInput, b as RubricPredictiveValidityReport, c as RubricRanking, r as rubricPredictiveValidity } from '../rubric-predictive-validity-
|
|
4
|
-
import { C as ContinuousAgreement, a as CalibrationResult, b as ContinuousCalibrationResult, c as CandidateScore, G as GoldenItem } from '../judge-calibration-
|
|
3
|
+
export { R as RubricOutcomePair, a as RubricPredictiveValidityInput, b as RubricPredictiveValidityReport, c as RubricRanking, r as rubricPredictiveValidity } from '../rubric-predictive-validity-ZGIlJNce.js';
|
|
4
|
+
import { C as ContinuousAgreement, a as CalibrationResult, b as ContinuousCalibrationResult, c as CandidateScore, G as GoldenItem } from '../judge-calibration-7C-IDmKr.js';
|
|
5
5
|
import { S as SeriesConvergenceOptions, a as SeriesConvergenceResult } from '../series-convergence-D5OWMBg6.js';
|
|
6
|
-
import { C as CorpusAgreementReport } from '../statistics-
|
|
6
|
+
import { C as CorpusAgreementReport } from '../statistics-D88peojY.js';
|
|
7
7
|
import '../store-BcFXE6LG.js';
|
|
8
8
|
import '../schema-m0gsnbt3.js';
|
|
9
|
-
import '../run-record-
|
|
9
|
+
import '../run-record-CPfd1ARZ.js';
|
|
10
10
|
import '@tangle-network/agent-interface';
|
|
11
11
|
import '../errors-CzMUYo7b.js';
|
|
12
12
|
import '../types-C7DGg5ex.js';
|
package/dist/meta-eval/index.js
CHANGED
|
@@ -11,15 +11,15 @@ import {
|
|
|
11
11
|
} from "../chunk-3RF76KTD.js";
|
|
12
12
|
import {
|
|
13
13
|
rubricPredictiveValidity
|
|
14
|
-
} from "../chunk-
|
|
14
|
+
} from "../chunk-QCBB6ZIU.js";
|
|
15
15
|
import {
|
|
16
16
|
pearsonR,
|
|
17
17
|
spearmanR
|
|
18
|
-
} from "../chunk-
|
|
18
|
+
} from "../chunk-HYGRFL7C.js";
|
|
19
19
|
import {
|
|
20
20
|
aggregateLlm,
|
|
21
21
|
llmSpans
|
|
22
|
-
} from "../chunk-
|
|
22
|
+
} from "../chunk-JZXGWLK5.js";
|
|
23
23
|
import "../chunk-5BKGXME7.js";
|
|
24
24
|
import {
|
|
25
25
|
ValidationError
|
|
@@ -138,6 +138,9 @@ declare function gradeSemanticStatus(input: {
|
|
|
138
138
|
available: boolean;
|
|
139
139
|
threshold?: number;
|
|
140
140
|
}): LayerStatus;
|
|
141
|
+
/**
|
|
142
|
+
* Ordered DAG of verification layers with dependency-based skipping, per-layer findings, soft-fail semantics, and a blended composite score across all passed layers.
|
|
143
|
+
*/
|
|
141
144
|
declare class MultiLayerVerifier<Env = unknown> {
|
|
142
145
|
private readonly layers;
|
|
143
146
|
constructor(layers: Layer<Env>[]);
|
|
@@ -1,7 +1,7 @@
|
|
|
1
|
-
import { J as JudgeScore } from '../types-
|
|
1
|
+
import { J as JudgeScore } from '../types-Bihq6-a3.js';
|
|
2
2
|
import { AgentProfile } from '@tangle-network/agent-interface';
|
|
3
3
|
import { M as MatrixResult } from '../types-BUxNaJ8c.js';
|
|
4
|
-
import '../run-record-
|
|
4
|
+
import '../run-record-CPfd1ARZ.js';
|
|
5
5
|
import '../errors-CzMUYo7b.js';
|
|
6
6
|
import '../schema-m0gsnbt3.js';
|
|
7
7
|
import '../verdict-C9MlYujm.js';
|
package/dist/openapi.json
CHANGED
|
@@ -2,7 +2,7 @@
|
|
|
2
2
|
"openapi": "3.1.0",
|
|
3
3
|
"info": {
|
|
4
4
|
"title": "@tangle-network/agent-eval — wire protocol",
|
|
5
|
-
"version": "0.
|
|
5
|
+
"version": "0.103.1",
|
|
6
6
|
"description": "HTTP and stdio RPC interface to agent-eval. The TypeScript runtime is the source of truth; this spec is the contract that cross-language clients (Python, Rust, Go) generate from.\n\nWire-protocol version: 1.0.0. Bumps on breaking changes to request/response schemas.",
|
|
7
7
|
"contact": {
|
|
8
8
|
"name": "Tangle Network",
|
|
@@ -4,7 +4,7 @@ export { a as FailureCluster, F as FailureClusterReport, f as failureClusterView
|
|
|
4
4
|
import { a as TrajectoryStep } from '../trajectory-2TkpSEVh.js';
|
|
5
5
|
import { B as BaselineOptions, a as BaselineReport } from '../baseline-Bbid3WoO.js';
|
|
6
6
|
export { c as computeToolUseMetrics } from '../baseline-Bbid3WoO.js';
|
|
7
|
-
import { l as llmSpans } from '../query-
|
|
7
|
+
import { l as llmSpans } from '../query-0aTmbmQe.js';
|
|
8
8
|
|
|
9
9
|
/**
|
|
10
10
|
* BudgetBreachView — aggregates breach events across the corpus.
|
package/dist/pipelines/index.js
CHANGED
|
@@ -3,21 +3,21 @@ import {
|
|
|
3
3
|
classifyFailure,
|
|
4
4
|
compareToBaseline,
|
|
5
5
|
computeToolUseMetrics
|
|
6
|
-
} from "../chunk-
|
|
6
|
+
} from "../chunk-BXZVBX3D.js";
|
|
7
7
|
import {
|
|
8
8
|
buildTrajectory
|
|
9
9
|
} from "../chunk-RZTMDUO7.js";
|
|
10
10
|
import {
|
|
11
11
|
interRaterReliability,
|
|
12
12
|
pearsonR
|
|
13
|
-
} from "../chunk-
|
|
13
|
+
} from "../chunk-HYGRFL7C.js";
|
|
14
14
|
import {
|
|
15
15
|
aggregateLlm,
|
|
16
16
|
argHash,
|
|
17
17
|
llmSpans,
|
|
18
18
|
runFailureClass,
|
|
19
19
|
toolSpans
|
|
20
|
-
} from "../chunk-
|
|
20
|
+
} from "../chunk-JZXGWLK5.js";
|
|
21
21
|
import "../chunk-5BKGXME7.js";
|
|
22
22
|
import "../chunk-3BFEG2F6.js";
|
|
23
23
|
import "../chunk-PZ5AY32C.js";
|
|
@@ -1,6 +1,6 @@
|
|
|
1
|
-
import { A as AgentProfileCell, a as AgentProfileJson } from './run-record-
|
|
1
|
+
import { A as AgentProfileCell, a as AgentProfileJson } from './run-record-CPfd1ARZ.js';
|
|
2
2
|
import { V as ValidationError } from './errors-CzMUYo7b.js';
|
|
3
|
-
import { A as AnalystFinding, E as EvidenceRef } from './types-
|
|
3
|
+
import { A as AnalystFinding, E as EvidenceRef } from './types-DA9yj-Jd.js';
|
|
4
4
|
|
|
5
5
|
type PolicyEditSchemaVersion = 'policy-edit/v1';
|
|
6
6
|
declare const POLICY_EDIT_AXES: readonly ["carrier", "representation", "budget", "sampling", "output_contract", "tool_contract", "routing", "memory", "agent_profile", "deployment_target"];
|
|
@@ -1,8 +1,9 @@
|
|
|
1
1
|
import { A as AgentEvalError } from './errors-CzMUYo7b.js';
|
|
2
|
-
import { R as RunRecord } from './run-record-
|
|
3
|
-
import { C as ChatClient } from './types-
|
|
4
|
-
import { c as JudgeDimension, S as Scenario, a as JudgeConfig } from './types-
|
|
2
|
+
import { R as RunRecord } from './run-record-CPfd1ARZ.js';
|
|
3
|
+
import { C as ChatClient } from './types-DA9yj-Jd.js';
|
|
4
|
+
import { c as JudgeDimension, S as Scenario, a as JudgeConfig } from './types-Bihq6-a3.js';
|
|
5
5
|
import { TCloud } from '@tangle-network/tcloud';
|
|
6
|
+
import { R as RawProviderSink } from './raw-provider-sink-C46HDghv.js';
|
|
6
7
|
import { D as DefaultVerdict } from './verdict-C9MlYujm.js';
|
|
7
8
|
|
|
8
9
|
/**
|
|
@@ -166,9 +167,10 @@ declare function containsAll(name: string, required: string[], options?: {
|
|
|
166
167
|
* actually fulfils the requirement. A hallucinated artifact fails here;
|
|
167
168
|
* an absent one already failed stage 1.
|
|
168
169
|
*
|
|
169
|
-
* `completionRate` is satisfied /
|
|
170
|
-
*
|
|
171
|
-
*
|
|
170
|
+
* `completionRate` is satisfied / MEASURABLE requirements (unmeasured rows —
|
|
171
|
+
* checker failures — are excluded from the denominator, never scored as
|
|
172
|
+
* zeros). Quality dimensions are meaningless on an incomplete task — callers
|
|
173
|
+
* gate on `fullyComplete` / `completionRate` before scoring quality.
|
|
172
174
|
*/
|
|
173
175
|
|
|
174
176
|
/** What kind of produced state can satisfy a requirement structurally. */
|
|
@@ -210,12 +212,23 @@ interface RequirementCheck {
|
|
|
210
212
|
structurallyPresent: boolean;
|
|
211
213
|
/**
|
|
212
214
|
* Whether the matched item actually fulfils the requirement. `null` when
|
|
213
|
-
* not structurally present,
|
|
214
|
-
* to assess.
|
|
215
|
+
* not structurally present, when the matched item carries no content
|
|
216
|
+
* to assess, or when the correctness check itself failed (`unmeasured`).
|
|
215
217
|
*/
|
|
216
218
|
correct: boolean | null;
|
|
217
|
-
/** structurallyPresent && correct !== false. */
|
|
219
|
+
/** structurallyPresent && !unmeasured && correct !== false. */
|
|
218
220
|
satisfied: boolean;
|
|
221
|
+
/**
|
|
222
|
+
* Set when the correctness check itself errored (LLM call failure or an
|
|
223
|
+
* unparseable response after retry). The requirement's fulfilment is
|
|
224
|
+
* UNKNOWN — `correct` stays null, `satisfied` is false, and
|
|
225
|
+
* `completionVerdict` excludes the row from `completionRate`'s
|
|
226
|
+
* denominator. Never folded into a zero: a synthetic zero is
|
|
227
|
+
* indistinguishable from a real failure (see `JudgeParseError`).
|
|
228
|
+
*/
|
|
229
|
+
unmeasured?: true;
|
|
230
|
+
/** Why the correctness check could not be measured (present iff `unmeasured`). */
|
|
231
|
+
unmeasuredReason?: string;
|
|
219
232
|
/** Human-readable evidence for the verdict. */
|
|
220
233
|
evidence: string[];
|
|
221
234
|
}
|
|
@@ -225,10 +238,12 @@ interface RequirementCheck {
|
|
|
225
238
|
interface CompletionVerdict extends DefaultVerdict {
|
|
226
239
|
taskId: string;
|
|
227
240
|
requirements: RequirementCheck[];
|
|
228
|
-
/** satisfied /
|
|
241
|
+
/** satisfied / MEASURABLE requirements (unmeasured rows leave the denominator). */
|
|
229
242
|
completionRate: number;
|
|
230
|
-
/** Every requirement satisfied. */
|
|
243
|
+
/** Every measurable requirement satisfied (false when anything is unmeasured). */
|
|
231
244
|
fullyComplete: boolean;
|
|
245
|
+
/** Requirements whose correctness check errored — reported, never scored as zero. */
|
|
246
|
+
unmeasuredCount: number;
|
|
232
247
|
}
|
|
233
248
|
/**
|
|
234
249
|
* Construct a `CompletionVerdict` from the per-requirement checks, deriving
|
|
@@ -262,8 +277,26 @@ interface LlmCorrectnessCheckerOpts {
|
|
|
262
277
|
model?: string;
|
|
263
278
|
/** Max chars of artifact content sent to the checker. */
|
|
264
279
|
maxContentChars?: number;
|
|
280
|
+
/**
|
|
281
|
+
* Checker LLM calls per requirement before giving up (parse failures and
|
|
282
|
+
* call errors both consume attempts). The failure then surfaces as an
|
|
283
|
+
* `unmeasured` requirement, never a zero.
|
|
284
|
+
*/
|
|
285
|
+
maxAttempts?: number;
|
|
286
|
+
/**
|
|
287
|
+
* Forensic capture of every checker request/response/error — without it a
|
|
288
|
+
* checker failure is unauditable (the agent-turn raws never contain the
|
|
289
|
+
* checker's own calls). Same sink contract as `LlmClient`.
|
|
290
|
+
*/
|
|
291
|
+
rawSink?: RawProviderSink;
|
|
265
292
|
}
|
|
266
|
-
/**
|
|
293
|
+
/**
|
|
294
|
+
* Parse the correctness checker's model response. Tolerates a response
|
|
295
|
+
* truncated mid-JSON (max_tokens cap) by auto-closing the prefix — the
|
|
296
|
+
* verdict boolean usually lands in the first few tokens, so a recovered
|
|
297
|
+
* prefix with a boolean `correct` is a real measurement, not a guess.
|
|
298
|
+
* Fails loud (JudgeParseError) when no boolean verdict is recoverable.
|
|
299
|
+
*/
|
|
267
300
|
declare function parseCorrectnessResponse(raw: string): {
|
|
268
301
|
correct: boolean;
|
|
269
302
|
reason: string;
|
|
@@ -1,3 +1,95 @@
|
|
|
1
|
+
import { R as RunRecord } from '../run-record-CPfd1ARZ.js';
|
|
2
|
+
import '@tangle-network/agent-interface';
|
|
3
|
+
import '../errors-CzMUYo7b.js';
|
|
4
|
+
import '../schema-m0gsnbt3.js';
|
|
5
|
+
|
|
6
|
+
/**
|
|
7
|
+
* Export side of the product benchmark bundle contract: convert product
|
|
8
|
+
* eval run directories (`records.jsonl` of `RunRecord` rows + trace/raw
|
|
9
|
+
* artifacts) into a portable `product-benchmark-manifest.json` +
|
|
10
|
+
* `product-benchmark-records.jsonl` bundle that
|
|
11
|
+
* `validateProductBenchmarkRun` accepts.
|
|
12
|
+
*
|
|
13
|
+
* Product-specific policy (safety-split detection, tool-call recovery,
|
|
14
|
+
* profile id fallback, artifact materialization) enters through explicit
|
|
15
|
+
* options; everything else is the shared union of the tax/legal/creative
|
|
16
|
+
* exporters. Scenario catalogs, smoke runners, and CLIs stay in the
|
|
17
|
+
* products.
|
|
18
|
+
*
|
|
19
|
+
* Input rows are checked structurally, not with `validateRunRecord`:
|
|
20
|
+
* product harnesses record bare model aliases and partial provenance, and
|
|
21
|
+
* the bundle contract's own validators re-check every field that matters
|
|
22
|
+
* on the way out.
|
|
23
|
+
*/
|
|
24
|
+
|
|
25
|
+
/** Full mutable-surface superset a product arm may declare. */
|
|
26
|
+
declare const productBenchmarkMutableSurfaces: readonly ["prompt", "resources.files", "tools", "mcp", "hooks", "subagents"];
|
|
27
|
+
interface ProductBenchmarkExportOptions {
|
|
28
|
+
/** Source eval run directories, each containing a `records.jsonl` of RunRecord rows. */
|
|
29
|
+
readonly runDirs: readonly string[];
|
|
30
|
+
/** Destination directory for the bundle (manifest + records + materialized source runs). */
|
|
31
|
+
readonly outDir: string;
|
|
32
|
+
readonly projectId: string;
|
|
33
|
+
readonly benchmarkId: string;
|
|
34
|
+
/** Repo-relative path of the product's canonical agent profile source. */
|
|
35
|
+
readonly agentProfilePath: string;
|
|
36
|
+
/** Pass threshold applied when a row carries no explicit `outcome.raw.pass`. Default 0.7. */
|
|
37
|
+
readonly passThreshold?: number;
|
|
38
|
+
/**
|
|
39
|
+
* First scenario tag. Defaults to `projectId` with a trailing `-agent`
|
|
40
|
+
* stripped (`tax-agent` → `tax`), matching the product exporters.
|
|
41
|
+
*/
|
|
42
|
+
readonly scenarioTagPrefix?: string;
|
|
43
|
+
/** Profile id used when a row has no `agentProfile.profileId`. Defaults to the row's arm id. */
|
|
44
|
+
readonly fallbackProfileId?: string;
|
|
45
|
+
/** Arm mutable surfaces recorded in the manifest. Defaults to the full superset. */
|
|
46
|
+
readonly mutableSurfaces?: readonly string[];
|
|
47
|
+
/**
|
|
48
|
+
* Copy each run dir into `<outDir>/source-runs/` and record
|
|
49
|
+
* bundle-relative artifact paths (portable, self-contained). When false,
|
|
50
|
+
* artifacts keep absolute paths into the original run dirs. Default true.
|
|
51
|
+
*/
|
|
52
|
+
readonly materializeSourceRuns?: boolean;
|
|
53
|
+
/**
|
|
54
|
+
* Override split classification for a row. Return undefined to fall back
|
|
55
|
+
* to the default (`outcome.raw.safety === 1` → safety, then splitTag).
|
|
56
|
+
*/
|
|
57
|
+
readonly classifySplit?: (record: RunRecord) => ProductBenchmarkSplit | undefined;
|
|
58
|
+
/** Recovers a tool-call count when the row's raw bag carries none (e.g. from turn artifacts). */
|
|
59
|
+
readonly toolCallFallback?: (record: RunRecord, runDir: string) => number;
|
|
60
|
+
/** Backend version recorded per row. Defaults to the cwd package.json's `@tangle-network/sandbox` range. */
|
|
61
|
+
readonly backendVersion?: string;
|
|
62
|
+
/**
|
|
63
|
+
* Explicit substrate versions for the manifest, merged over what the cwd
|
|
64
|
+
* package.json / node_modules resolve. Use when a substrate package is not
|
|
65
|
+
* installed where the export runs — the validator refuses an `'unknown'`
|
|
66
|
+
* version, so provide the real one rather than shipping the sentinel.
|
|
67
|
+
*/
|
|
68
|
+
readonly substrate?: Partial<ProductBenchmarkManifest['substrate']>;
|
|
69
|
+
}
|
|
70
|
+
interface ProductBenchmarkSingleRunExportOptions extends Omit<ProductBenchmarkExportOptions, 'runDirs'> {
|
|
71
|
+
readonly runDir: string;
|
|
72
|
+
}
|
|
73
|
+
interface ProductBenchmarkExportResult {
|
|
74
|
+
readonly manifestPath: string;
|
|
75
|
+
readonly recordsPath: string;
|
|
76
|
+
readonly records: number;
|
|
77
|
+
}
|
|
78
|
+
/** Repo identity from the exporting process's cwd. `'unknown'` values are flagged by `validateProductBenchmarkRun`. */
|
|
79
|
+
declare function productBenchmarkRepoIdentity(): ProductBenchmarkManifest['repo'];
|
|
80
|
+
/** Map one RunRecord row to a validated product benchmark record. */
|
|
81
|
+
declare function runRecordToProductBenchmarkRecord(record: RunRecord, runDir: string, artifactRoot: string, artifacts: ProductBenchmarkRecord['artifacts'], options: ProductBenchmarkExportOptions | ProductBenchmarkSingleRunExportOptions): ProductBenchmarkRecord;
|
|
82
|
+
/** Derive the bundle manifest from already-normalized records. */
|
|
83
|
+
declare function buildProductBenchmarkManifest(records: readonly ProductBenchmarkRecord[], options: Pick<ProductBenchmarkExportOptions, 'outDir' | 'projectId' | 'benchmarkId' | 'scenarioTagPrefix' | 'mutableSurfaces' | 'substrate'>): ProductBenchmarkManifest;
|
|
84
|
+
/** Single-run convenience wrapper over `exportProductBenchmarkRuns`. */
|
|
85
|
+
declare function exportProductBenchmark(options: ProductBenchmarkSingleRunExportOptions): ProductBenchmarkExportResult;
|
|
86
|
+
/**
|
|
87
|
+
* Export one or more product eval run dirs into a validated product
|
|
88
|
+
* benchmark bundle at `outDir`. Both the manifest and every record are
|
|
89
|
+
* run through the contract validators before anything is written.
|
|
90
|
+
*/
|
|
91
|
+
declare function exportProductBenchmarkRuns(options: ProductBenchmarkExportOptions): ProductBenchmarkExportResult;
|
|
92
|
+
|
|
1
93
|
declare const productBenchmarkSplits: readonly ["practice", "dev", "holdout", "safety", "sentinel"];
|
|
2
94
|
type ProductBenchmarkSplit = (typeof productBenchmarkSplits)[number];
|
|
3
95
|
interface ProductBenchmarkRepoRef {
|
|
@@ -116,6 +208,11 @@ interface ProductBenchmarkValidationReport {
|
|
|
116
208
|
readonly manifestPath: string;
|
|
117
209
|
readonly recordsPath: string;
|
|
118
210
|
readonly records: number;
|
|
211
|
+
/** Manifest repo fields that are empty or the `'unknown'` export sentinel. */
|
|
212
|
+
readonly repoFailures: readonly string[];
|
|
213
|
+
/** Manifest substrate versions that are empty or the `'unknown'` export
|
|
214
|
+
* sentinel — a bundle without substrate identity is not reproducible. */
|
|
215
|
+
readonly substrateFailures: readonly string[];
|
|
119
216
|
readonly projects: readonly string[];
|
|
120
217
|
readonly benchmarks: readonly string[];
|
|
121
218
|
readonly arms: readonly string[];
|
|
@@ -140,5 +237,11 @@ declare function readProductBenchmarkRecords(path: string): ProductBenchmarkReco
|
|
|
140
237
|
declare function readProductBenchmarkManifest(path: string): ProductBenchmarkManifest;
|
|
141
238
|
declare function validateProductBenchmarkRun(input: ProductBenchmarkRunInput): ProductBenchmarkValidationReport;
|
|
142
239
|
declare function findProductBenchmarkArtifacts(runDir: string): ProductBenchmarkArtifactPaths | null;
|
|
240
|
+
/**
|
|
241
|
+
* Fail-loud gate over a bundle directory: locates the manifest + records,
|
|
242
|
+
* runs `validateProductBenchmarkRun`, and throws with every repo,
|
|
243
|
+
* integrity, and artifact failure listed. Returns the report when clean.
|
|
244
|
+
*/
|
|
245
|
+
declare function assertProductBenchmarkRun(runDir: string): ProductBenchmarkValidationReport;
|
|
143
246
|
|
|
144
|
-
export { type AgentProfileRuntimeReceipt, type ProductBenchmarkArm, type ProductBenchmarkArtifactPaths, type ProductBenchmarkBudgets, type ProductBenchmarkManifest, type ProductBenchmarkProfileRef, type ProductBenchmarkRecord, type ProductBenchmarkRepoRef, type ProductBenchmarkRunInput, type ProductBenchmarkScenario, type ProductBenchmarkSplit, type ProductBenchmarkSubstrateVersions, type ProductBenchmarkValidationReport, type RuntimeResolution, findProductBenchmarkArtifacts, productBenchmarkIntegrityFailures, productBenchmarkSplits, readProductBenchmarkManifest, readProductBenchmarkRecords, validateProductBenchmarkManifest, validateProductBenchmarkRecord, validateProductBenchmarkRun };
|
|
247
|
+
export { type AgentProfileRuntimeReceipt, type ProductBenchmarkArm, type ProductBenchmarkArtifactPaths, type ProductBenchmarkBudgets, type ProductBenchmarkExportOptions, type ProductBenchmarkExportResult, type ProductBenchmarkManifest, type ProductBenchmarkProfileRef, type ProductBenchmarkRecord, type ProductBenchmarkRepoRef, type ProductBenchmarkRunInput, type ProductBenchmarkScenario, type ProductBenchmarkSingleRunExportOptions, type ProductBenchmarkSplit, type ProductBenchmarkSubstrateVersions, type ProductBenchmarkValidationReport, type RuntimeResolution, assertProductBenchmarkRun, buildProductBenchmarkManifest, exportProductBenchmark, exportProductBenchmarkRuns, findProductBenchmarkArtifacts, productBenchmarkIntegrityFailures, productBenchmarkMutableSurfaces, productBenchmarkRepoIdentity, productBenchmarkSplits, readProductBenchmarkManifest, readProductBenchmarkRecords, runRecordToProductBenchmarkRecord, validateProductBenchmarkManifest, validateProductBenchmarkRecord, validateProductBenchmarkRun };
|
|
@@ -1,21 +1,35 @@
|
|
|
1
1
|
import {
|
|
2
|
+
assertProductBenchmarkRun,
|
|
3
|
+
buildProductBenchmarkManifest,
|
|
4
|
+
exportProductBenchmark,
|
|
5
|
+
exportProductBenchmarkRuns,
|
|
2
6
|
findProductBenchmarkArtifacts,
|
|
3
7
|
productBenchmarkIntegrityFailures,
|
|
8
|
+
productBenchmarkMutableSurfaces,
|
|
9
|
+
productBenchmarkRepoIdentity,
|
|
4
10
|
productBenchmarkSplits,
|
|
5
11
|
readProductBenchmarkManifest,
|
|
6
12
|
readProductBenchmarkRecords,
|
|
13
|
+
runRecordToProductBenchmarkRecord,
|
|
7
14
|
validateProductBenchmarkManifest,
|
|
8
15
|
validateProductBenchmarkRecord,
|
|
9
16
|
validateProductBenchmarkRun
|
|
10
|
-
} from "../chunk-
|
|
17
|
+
} from "../chunk-RQNOLV3I.js";
|
|
11
18
|
import "../chunk-3BFEG2F6.js";
|
|
12
19
|
import "../chunk-PZ5AY32C.js";
|
|
13
20
|
export {
|
|
21
|
+
assertProductBenchmarkRun,
|
|
22
|
+
buildProductBenchmarkManifest,
|
|
23
|
+
exportProductBenchmark,
|
|
24
|
+
exportProductBenchmarkRuns,
|
|
14
25
|
findProductBenchmarkArtifacts,
|
|
15
26
|
productBenchmarkIntegrityFailures,
|
|
27
|
+
productBenchmarkMutableSurfaces,
|
|
28
|
+
productBenchmarkRepoIdentity,
|
|
16
29
|
productBenchmarkSplits,
|
|
17
30
|
readProductBenchmarkManifest,
|
|
18
31
|
readProductBenchmarkRecords,
|
|
32
|
+
runRecordToProductBenchmarkRecord,
|
|
19
33
|
validateProductBenchmarkManifest,
|
|
20
34
|
validateProductBenchmarkRecord,
|
|
21
35
|
validateProductBenchmarkRun
|
|
@@ -1,9 +1,9 @@
|
|
|
1
|
-
import { S as Scenario, g as Gate, G as GateResult, h as GateContext, C as CampaignResult, i as Mutator, f as SurfaceProposer, M as MutableSurface, j as GateDecision } from './types-
|
|
1
|
+
import { S as Scenario, g as Gate, G as GateResult, h as GateContext, C as CampaignResult, i as Mutator, f as SurfaceProposer, M as MutableSurface, j as GateDecision } from './types-Bihq6-a3.js';
|
|
2
2
|
import { R as RedTeamCase } from './red-team-BWdoyleI.js';
|
|
3
|
-
import { R as RunRecord } from './run-record-
|
|
3
|
+
import { R as RunRecord } from './run-record-CPfd1ARZ.js';
|
|
4
4
|
import { D as Direction } from './pareto-E-pembql.js';
|
|
5
|
-
import { a as PairedBootstrapResult } from './statistics-
|
|
6
|
-
import { R as RunCampaignOptions, C as CampaignStorage } from './gepa-
|
|
5
|
+
import { a as PairedBootstrapResult } from './statistics-D88peojY.js';
|
|
6
|
+
import { R as RunCampaignOptions, C as CampaignStorage } from './gepa-COlCAkHN.js';
|
|
7
7
|
import { HostedClient, TraceSpanEvent } from './hosted/index.js';
|
|
8
8
|
|
|
9
9
|
/**
|
|
@@ -72,9 +72,13 @@ interface DefaultProductionGateOptions {
|
|
|
72
72
|
* fires at the `gaming` severity. Default true. */
|
|
73
73
|
blockOnRewardHackingGaming?: boolean;
|
|
74
74
|
}
|
|
75
|
+
/**
|
|
76
|
+
* Opinionated production gate composing held-out significance, red-team, reward-hacking, and canary checks into a single `Gate.decide` decision.
|
|
77
|
+
*/
|
|
75
78
|
declare function defaultProductionGate<TArtifact, TScenario extends Scenario>(options: DefaultProductionGateOptions): Gate<TArtifact, TScenario>;
|
|
76
79
|
|
|
77
80
|
/**
|
|
81
|
+
* @module
|
|
78
82
|
* Thin Gate adapter — exposes delta-threshold-on-holdout as a composable
|
|
79
83
|
* `Gate`. Use when you want held-out as one of N composed gates instead of
|
|
80
84
|
* the full `defaultProductionGate` stack.
|
|
@@ -84,6 +88,9 @@ interface HeldOutGateOptions<TScenario extends Scenario = Scenario> {
|
|
|
84
88
|
scenarios: TScenario[];
|
|
85
89
|
deltaThreshold?: number;
|
|
86
90
|
}
|
|
91
|
+
/**
|
|
92
|
+
* Composable held-out delta gate: ships only when the candidate's mean composite on `scenarios` beats the baseline by at least `deltaThreshold`.
|
|
93
|
+
*/
|
|
87
94
|
declare function heldOutGate<TArtifact, TScenario extends Scenario>(options: HeldOutGateOptions<TScenario>): Gate<TArtifact, TScenario>;
|
|
88
95
|
|
|
89
96
|
/**
|
|
@@ -220,6 +227,9 @@ declare function paretoSignificanceGate<TArtifact = unknown, TScenario extends S
|
|
|
220
227
|
interface RunEvalOptions<TScenario extends Scenario, TArtifact> extends Omit<RunCampaignOptions<TScenario, TArtifact>, 'runDir'> {
|
|
221
228
|
runDir: string;
|
|
222
229
|
}
|
|
230
|
+
/**
|
|
231
|
+
* Simplest evaluation preset: run scenarios through dispatch, score with judges, and return a `CampaignResult` — no optimizer, no gate, no PR.
|
|
232
|
+
*/
|
|
223
233
|
declare function runEval<TScenario extends Scenario, TArtifact>(opts: RunEvalOptions<TScenario, TArtifact>): Promise<CampaignResult<TArtifact, TScenario>>;
|
|
224
234
|
|
|
225
235
|
/**
|
|
@@ -240,6 +250,9 @@ interface EvolutionaryProposerOptions<TFindings = unknown> {
|
|
|
240
250
|
/** External findings fed to the mutator each generation. Default: []. */
|
|
241
251
|
findings?: TFindings[];
|
|
242
252
|
}
|
|
253
|
+
/**
|
|
254
|
+
* Wrap a stateless `Mutator` (GEPA, AxGEPA, reflective-mutation) as a `SurfaceProposer` that mutates the current best surface into N candidates each generation.
|
|
255
|
+
*/
|
|
243
256
|
declare function evolutionaryProposer<TFindings = unknown>(opts: EvolutionaryProposerOptions<TFindings>): SurfaceProposer<TFindings>;
|
|
244
257
|
|
|
245
258
|
/**
|
|
@@ -396,6 +409,9 @@ declare function loopProvenanceSpans(record: LoopProvenanceRecord, opts?: {
|
|
|
396
409
|
}): TraceSpanEvent[];
|
|
397
410
|
/** Canonical durable paths under the run dir. */
|
|
398
411
|
declare function provenanceRecordPath(runDir: string): string;
|
|
412
|
+
/**
|
|
413
|
+
* Canonical path for the durable OTLP spans JSONL file under a loop run directory.
|
|
414
|
+
*/
|
|
399
415
|
declare function provenanceSpansPath(runDir: string): string;
|
|
400
416
|
interface EmitLoopProvenanceResult {
|
|
401
417
|
record: LoopProvenanceRecord;
|
|
@@ -13,6 +13,7 @@ import { T as TraceStore } from './store-BcFXE6LG.js';
|
|
|
13
13
|
declare function runsForScenario(store: TraceStore, scenarioId: string): Promise<Run[]>;
|
|
14
14
|
declare function llmSpans(store: TraceStore, runId?: string): Promise<LlmSpan[]>;
|
|
15
15
|
declare function toolSpans(store: TraceStore, runId?: string, toolName?: string): Promise<ToolSpan[]>;
|
|
16
|
+
/** Query judge-kind spans from the trace store, optionally scoped to a single run. */
|
|
16
17
|
declare function judgeSpans(store: TraceStore, runId?: string): Promise<JudgeSpan[]>;
|
|
17
18
|
/** Group spans by any key selector. */
|
|
18
19
|
declare function groupBy<T, K extends string | number>(items: T[], key: (t: T) => K): Map<K, T[]>;
|
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
import { D as DatasetSplit, c as DatasetManifest, a as DatasetScenario } from './dataset-BbGkaN2I.js';
|
|
2
|
-
import { m as GateDecision } from './summary-report-
|
|
3
|
-
import { R as RunRecord, b as RunSplitTag } from './run-record-
|
|
2
|
+
import { m as GateDecision } from './summary-report-Fc_YFJat.js';
|
|
3
|
+
import { R as RunRecord, b as RunSplitTag } from './run-record-CPfd1ARZ.js';
|
|
4
4
|
|
|
5
5
|
/**
|
|
6
6
|
* Release confidence gate.
|
|
@@ -216,6 +216,9 @@ interface JudgeReplayGateArgs<TOutput> {
|
|
|
216
216
|
/** Maximum concurrent judge calls. Default 4. */
|
|
217
217
|
judgeConcurrency?: number;
|
|
218
218
|
}
|
|
219
|
+
/**
|
|
220
|
+
* Confirm a candidate's win with a stronger judge: score baseline and candidate outputs independently, then bootstrap a CI to verify the lift generalises beyond the inner loop.
|
|
221
|
+
*/
|
|
219
222
|
declare function judgeReplayGate<TOutput>(args: JudgeReplayGateArgs<TOutput>): Promise<BootstrapResult & {
|
|
220
223
|
baselineSamples: number;
|
|
221
224
|
candidateSamples: number;
|
package/dist/reporting.d.ts
CHANGED
|
@@ -1,15 +1,15 @@
|
|
|
1
|
-
export { R as RubricOutcomePair, a as RubricPredictiveValidityInput, b as RubricPredictiveValidityReport, c as RubricRanking, r as rubricPredictiveValidity } from './rubric-predictive-validity-
|
|
2
|
-
export { B as BootstrapOptions, a as BootstrapResult, J as JudgeReplayGateArgs, R as ReleaseConfidenceAxis, b as ReleaseConfidenceAxisName, c as ReleaseConfidenceInput, d as ReleaseConfidenceIssue, e as ReleaseConfidenceMetrics, f as ReleaseConfidenceScorecard, g as ReleaseConfidenceStatus, h as ReleaseConfidenceThresholds, i as ReleaseTraceEvidence, j as RenderReleaseReportOptions, V as Verdict, k as assertReleaseConfidence, l as bootstrapCi, m as evaluateReleaseConfidence, n as judgeReplayGate, r as renderReleaseReport } from './release-report-
|
|
1
|
+
export { R as RubricOutcomePair, a as RubricPredictiveValidityInput, b as RubricPredictiveValidityReport, c as RubricRanking, r as rubricPredictiveValidity } from './rubric-predictive-validity-ZGIlJNce.js';
|
|
2
|
+
export { B as BootstrapOptions, a as BootstrapResult, J as JudgeReplayGateArgs, R as ReleaseConfidenceAxis, b as ReleaseConfidenceAxisName, c as ReleaseConfidenceInput, d as ReleaseConfidenceIssue, e as ReleaseConfidenceMetrics, f as ReleaseConfidenceScorecard, g as ReleaseConfidenceStatus, h as ReleaseConfidenceThresholds, i as ReleaseTraceEvidence, j as RenderReleaseReportOptions, V as Verdict, k as assertReleaseConfidence, l as bootstrapCi, m as evaluateReleaseConfidence, n as judgeReplayGate, r as renderReleaseReport } from './release-report-DkaCZ9k4.js';
|
|
3
3
|
export { I as InterimReleaseConfidence, a as InterimReleaseConfidenceInput, P as PairedEvalueOptions, b as PairedEvalueSequence, c as PairedEvalueStep, S as SequentialDecision, e as evaluateInterimReleaseConfidence, p as pairedEvalueSequence } from './sequential-5iSVfzl2.js';
|
|
4
|
-
export { P as PairedBootstrapOptions, a as PairedBootstrapResult, b as benjaminiHochberg, p as pairedBootstrap, w as wilcoxonSignedRank } from './statistics-
|
|
5
|
-
export { G as GainDistributionBin, a as GainDistributionFigureSpec, b as GainDistributionOptions, P as ParetoFigureSpec, c as ParetoPoint, R as RESEARCH_REPORT_HARD_PAIR_FLOOR, d as ResearchReport, e as ResearchReportCandidate, f as ResearchReportDecision, g as ResearchReportMethodology, h as ResearchReportOptions, i as ResearchReportRecommendation, S as SummaryTable, j as SummaryTableOptions, k as SummaryTableRow, l as gainHistogram, p as paretoChart, r as researchReport, s as summaryTable } from './summary-report-
|
|
6
|
-
import './run-record-
|
|
4
|
+
export { P as PairedBootstrapOptions, a as PairedBootstrapResult, b as benjaminiHochberg, p as pairedBootstrap, w as wilcoxonSignedRank } from './statistics-D88peojY.js';
|
|
5
|
+
export { G as GainDistributionBin, a as GainDistributionFigureSpec, b as GainDistributionOptions, P as ParetoFigureSpec, c as ParetoPoint, R as RESEARCH_REPORT_HARD_PAIR_FLOOR, d as ResearchReport, e as ResearchReportCandidate, f as ResearchReportDecision, g as ResearchReportMethodology, h as ResearchReportOptions, i as ResearchReportRecommendation, S as SummaryTable, j as SummaryTableOptions, k as SummaryTableRow, l as gainHistogram, p as paretoChart, r as researchReport, s as summaryTable } from './summary-report-Fc_YFJat.js';
|
|
6
|
+
import './run-record-CPfd1ARZ.js';
|
|
7
7
|
import '@tangle-network/agent-interface';
|
|
8
8
|
import './errors-CzMUYo7b.js';
|
|
9
9
|
import './schema-m0gsnbt3.js';
|
|
10
10
|
import './outcome-store-rnXLEqSn.js';
|
|
11
11
|
import './dataset-BbGkaN2I.js';
|
|
12
|
-
import './judge-calibration-
|
|
12
|
+
import './judge-calibration-7C-IDmKr.js';
|
|
13
13
|
import './types-C7DGg5ex.js';
|
|
14
14
|
import '@tangle-network/tcloud';
|
|
15
15
|
import './failure-cluster-DH9Flgcf.js';
|
package/dist/reporting.js
CHANGED
|
@@ -4,10 +4,10 @@ import {
|
|
|
4
4
|
evaluateReleaseConfidence,
|
|
5
5
|
judgeReplayGate,
|
|
6
6
|
renderReleaseReport
|
|
7
|
-
} from "./chunk-
|
|
7
|
+
} from "./chunk-UMEAR2FI.js";
|
|
8
8
|
import {
|
|
9
9
|
rubricPredictiveValidity
|
|
10
|
-
} from "./chunk-
|
|
10
|
+
} from "./chunk-QCBB6ZIU.js";
|
|
11
11
|
import {
|
|
12
12
|
evaluateInterimReleaseConfidence,
|
|
13
13
|
pairedEvalueSequence
|
|
@@ -18,12 +18,12 @@ import {
|
|
|
18
18
|
paretoChart,
|
|
19
19
|
researchReport,
|
|
20
20
|
summaryTable
|
|
21
|
-
} from "./chunk-
|
|
21
|
+
} from "./chunk-CLS3374R.js";
|
|
22
22
|
import {
|
|
23
23
|
benjaminiHochberg,
|
|
24
24
|
pairedBootstrap,
|
|
25
25
|
wilcoxonSignedRank
|
|
26
|
-
} from "./chunk-
|
|
26
|
+
} from "./chunk-HYGRFL7C.js";
|
|
27
27
|
import "./chunk-VSMTAMNK.js";
|
|
28
28
|
import "./chunk-3BFEG2F6.js";
|
|
29
29
|
import "./chunk-PZ5AY32C.js";
|
|
@@ -1,6 +1,6 @@
|
|
|
1
|
-
import { b as RunSplitTag, c as RunTokenUsage, d as RunJudgeMetadata, J as JudgeScoresRecord, A as AgentProfileCell, e as AgentProfileCellInput, R as RunRecord } from './run-record-
|
|
1
|
+
import { b as RunSplitTag, c as RunTokenUsage, d as RunJudgeMetadata, J as JudgeScoresRecord, A as AgentProfileCell, e as AgentProfileCellInput, R as RunRecord } from './run-record-CPfd1ARZ.js';
|
|
2
2
|
import { L as LlmClientOptions, a as LlmRouteRequirements } from './llm-client-Bj7g0rqu.js';
|
|
3
|
-
import { h as ResearchReportOptions, d as ResearchReport, m as GateDecision } from './summary-report-
|
|
3
|
+
import { h as ResearchReportOptions, d as ResearchReport, m as GateDecision } from './summary-report-Fc_YFJat.js';
|
|
4
4
|
import { T as TraceEmitter, R as RunCompleteHook } from './emitter-C2rqGH_l.js';
|
|
5
5
|
import { R as RunIntegrityExpectations, a as RunIntegrityReport } from './integrity-D2t12mMw.js';
|
|
6
6
|
import { R as RawProviderSink } from './raw-provider-sink-C46HDghv.js';
|
package/dist/rl.d.ts
CHANGED
|
@@ -1,23 +1,23 @@
|
|
|
1
|
-
import { R as RunRecord, b as RunSplitTag } from './run-record-
|
|
1
|
+
import { R as RunRecord, b as RunSplitTag } from './run-record-CPfd1ARZ.js';
|
|
2
2
|
export { A as AdversarialMutation } from './adversarial-B7loGVVX.js';
|
|
3
|
-
import { P as PreferenceExtractionReport, D as DpoExportRow, G as GrpoExportRow, S as SftExportRow, E as ExtractPreferencesOptions, a as DpoLookups, b as GrpoLookups, c as SftLookups } from './corpus-
|
|
4
|
-
export { d as CorpusAppendResult, C as CorpusRecord, e as DatasetFormat, f as ExtractStepRewardsOptions, H as HarvestOptions, g as PreferenceStrategy, h as PreferenceTriple, i as PrmExportRow, j as PrmLookups, k as PrmTrainingTriple, R as RewardKind, l as RewardStats, m as RlDatasetBundle, n as RlDatasetConfig, o as RlDatasetManifest, p as RlDatasetStats, q as RunwiseStepSummary, r as StepReward, s as StepRewardJsonlRow, t as StepScorer, u as appendToCorpus, v as buildDatasetFromCorpus, w as buildRlDataset, x as datasheetToMarkdown, y as extractPreferences, z as extractStepRewards, A as prmTrainingPairs, B as readCorpus, F as runwiseStepRewardSummary, I as stepRewardsToJsonl, J as toAnthropicFormat, K as toDpoJsonl, L as toDpoRows, M as toGrpoJsonl, N as toGrpoRows, O as toPrmJsonl, Q as toPrmRows, T as toSftJsonl, U as toSftRows, V as toTRLFormat } from './corpus-
|
|
3
|
+
import { P as PreferenceExtractionReport, D as DpoExportRow, G as GrpoExportRow, S as SftExportRow, E as ExtractPreferencesOptions, a as DpoLookups, b as GrpoLookups, c as SftLookups } from './corpus-CysLCiK4.js';
|
|
4
|
+
export { d as CorpusAppendResult, C as CorpusRecord, e as DatasetFormat, f as ExtractStepRewardsOptions, H as HarvestOptions, g as PreferenceStrategy, h as PreferenceTriple, i as PrmExportRow, j as PrmLookups, k as PrmTrainingTriple, R as RewardKind, l as RewardStats, m as RlDatasetBundle, n as RlDatasetConfig, o as RlDatasetManifest, p as RlDatasetStats, q as RunwiseStepSummary, r as StepReward, s as StepRewardJsonlRow, t as StepScorer, u as appendToCorpus, v as buildDatasetFromCorpus, w as buildRlDataset, x as datasheetToMarkdown, y as extractPreferences, z as extractStepRewards, A as prmTrainingPairs, B as readCorpus, F as runwiseStepRewardSummary, I as stepRewardsToJsonl, J as toAnthropicFormat, K as toDpoJsonl, L as toDpoRows, M as toGrpoJsonl, N as toGrpoRows, O as toPrmJsonl, Q as toPrmRows, T as toSftJsonl, U as toSftRows, V as toTRLFormat } from './corpus-CysLCiK4.js';
|
|
5
5
|
export { O as OffPolicyEstimate, a as OffPolicyOptions, b as OffPolicyTrajectory, d as doublyRobust, i as inverseProbabilityWeighting, o as offPolicyEstimateAll, s as selfNormalizedImportanceWeighting } from './off-policy-DiwuKKg7.js';
|
|
6
6
|
import { b as OutcomeStore } from './outcome-store-rnXLEqSn.js';
|
|
7
7
|
export { D as DeploymentOutcome, F as FileSystemOutcomeStore, a as FileSystemOutcomeStoreOptions, I as InMemoryOutcomeStore } from './outcome-store-rnXLEqSn.js';
|
|
8
|
-
import { b as RubricPredictiveValidityReport } from './rubric-predictive-validity-
|
|
9
|
-
import { R as Researcher, F as FailureMode, S as SteeringChange, E as ExperimentPlan, a as ExperimentResult, b as EvalCampaignResult, c as EvalCampaignOptions } from './researcher-
|
|
10
|
-
export { r as runEvalCampaign } from './researcher-
|
|
11
|
-
import { a as VerificationReport } from './multi-layer-verifier-
|
|
8
|
+
import { b as RubricPredictiveValidityReport } from './rubric-predictive-validity-ZGIlJNce.js';
|
|
9
|
+
import { R as Researcher, F as FailureMode, S as SteeringChange, E as ExperimentPlan, a as ExperimentResult, b as EvalCampaignResult, c as EvalCampaignOptions } from './researcher-SDAezfML.js';
|
|
10
|
+
export { r as runEvalCampaign } from './researcher-SDAezfML.js';
|
|
11
|
+
import { a as VerificationReport } from './multi-layer-verifier-CI4jdX-q.js';
|
|
12
12
|
import { I as InterimReleaseConfidence } from './sequential-5iSVfzl2.js';
|
|
13
|
-
import { C as CampaignResult } from './types-
|
|
13
|
+
import { C as CampaignResult } from './types-Bihq6-a3.js';
|
|
14
14
|
import '@tangle-network/agent-interface';
|
|
15
15
|
import './errors-CzMUYo7b.js';
|
|
16
16
|
import './schema-m0gsnbt3.js';
|
|
17
17
|
import './store-BcFXE6LG.js';
|
|
18
18
|
import './llm-client-Bj7g0rqu.js';
|
|
19
19
|
import './raw-provider-sink-C46HDghv.js';
|
|
20
|
-
import './summary-report-
|
|
20
|
+
import './summary-report-Fc_YFJat.js';
|
|
21
21
|
import './failure-cluster-DH9Flgcf.js';
|
|
22
22
|
import './emitter-C2rqGH_l.js';
|
|
23
23
|
import './integrity-D2t12mMw.js';
|