@tangle-network/agent-eval 0.112.0 → 0.114.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +34 -0
- package/dist/belief-state/index.js +1 -1
- package/dist/benchmarks/index.d.ts +2 -2
- package/dist/benchmarks/index.js +6 -6
- package/dist/builder-eval/index.js +1 -1
- package/dist/campaign/index.d.ts +55 -22
- package/dist/campaign/index.js +13 -7
- package/dist/{chunk-VLNGPUJJ.js → chunk-3LXTCTWL.js} +2 -2
- package/dist/{chunk-7QPK6W7R.js → chunk-ARU2PZFM.js} +2 -2
- package/dist/{chunk-EMKORATZ.js → chunk-DPZAEKA6.js} +2 -2
- package/dist/{chunk-XNCF3JU3.js → chunk-FAOEFFRT.js} +2 -2
- package/dist/{chunk-WZOIFFG2.js → chunk-LOBMT6SB.js} +2 -2
- package/dist/{chunk-RT2AFUXM.js → chunk-MOXWMGPC.js} +2 -2
- package/dist/{chunk-RCXMWGRY.js → chunk-N6MTC3GK.js} +69 -25
- package/dist/chunk-N6MTC3GK.js.map +1 -0
- package/dist/{chunk-YWGKQARF.js → chunk-NYFUT3B3.js} +2 -2
- package/dist/{chunk-VC43KQCK.js → chunk-PJQFMIOX.js} +62 -9
- package/dist/chunk-PJQFMIOX.js.map +1 -0
- package/dist/{chunk-QQVYMBCZ.js → chunk-VHPD6AXX.js} +512 -21
- package/dist/{chunk-QQVYMBCZ.js.map → chunk-VHPD6AXX.js.map} +1 -1
- package/dist/{chunk-BOLYPSAE.js → chunk-WBOGKYM4.js} +4 -4
- package/dist/{chunk-3LDPUYAC.js → chunk-X4UCIOTZ.js} +2 -2
- package/dist/contract/index.d.ts +7 -7
- package/dist/contract/index.js +9 -8
- package/dist/contract/index.js.map +1 -1
- package/dist/{gepa-T8T215nw.d.ts → gepa-DolL_Fko.d.ts} +2 -6
- package/dist/hosted/index.d.ts +1 -1
- package/dist/{index-Dc3VLGhp.d.ts → index-CWr5SIG-.d.ts} +1 -1
- package/dist/index.d.ts +8 -8
- package/dist/index.js +13 -11
- package/dist/index.js.map +1 -1
- package/dist/meta-eval/index.d.ts +1 -1
- package/dist/meta-eval/index.js +2 -2
- package/dist/multishot/index.d.ts +1 -1
- package/dist/openapi.json +1 -1
- package/dist/pipelines/index.js +2 -2
- package/dist/{pre-registration-BepVVa6P.d.ts → pre-registration-CTQbZbpX.d.ts} +1 -1
- package/dist/{provenance-KhY8ESVM.d.ts → provenance-BZmpWmn4.d.ts} +5 -11
- package/dist/reporting.d.ts +1 -1
- package/dist/reporting.js +4 -4
- package/dist/rl.d.ts +1 -1
- package/dist/rl.js +5 -5
- package/dist/{run-campaign-HJQEDCJQ.js → run-campaign-UADIM77S.js} +3 -3
- package/dist/{statistics-CDfpwIdX.d.ts → statistics-oUbOJe-S.d.ts} +34 -1
- package/dist/{types-v--ctu-b.d.ts → types-CgSlO6wT.d.ts} +26 -11
- package/docs/design/loop-taxonomy.md +7 -1
- package/docs/improvement-glossary.md +1 -1
- package/package.json +1 -1
- package/dist/chunk-RCXMWGRY.js.map +0 -1
- package/dist/chunk-VC43KQCK.js.map +0 -1
- /package/dist/{chunk-VLNGPUJJ.js.map → chunk-3LXTCTWL.js.map} +0 -0
- /package/dist/{chunk-7QPK6W7R.js.map → chunk-ARU2PZFM.js.map} +0 -0
- /package/dist/{chunk-EMKORATZ.js.map → chunk-DPZAEKA6.js.map} +0 -0
- /package/dist/{chunk-XNCF3JU3.js.map → chunk-FAOEFFRT.js.map} +0 -0
- /package/dist/{chunk-WZOIFFG2.js.map → chunk-LOBMT6SB.js.map} +0 -0
- /package/dist/{chunk-RT2AFUXM.js.map → chunk-MOXWMGPC.js.map} +0 -0
- /package/dist/{chunk-YWGKQARF.js.map → chunk-NYFUT3B3.js.map} +0 -0
- /package/dist/{chunk-BOLYPSAE.js.map → chunk-WBOGKYM4.js.map} +0 -0
- /package/dist/{chunk-3LDPUYAC.js.map → chunk-X4UCIOTZ.js.map} +0 -0
- /package/dist/{run-campaign-HJQEDCJQ.js.map → run-campaign-UADIM77S.js.map} +0 -0
|
@@ -3,7 +3,7 @@ export { D as DeploymentOutcome, F as FileSystemOutcomeStore, a as FileSystemOut
|
|
|
3
3
|
export { R as RubricOutcomePair, a as RubricPredictiveValidityInput, b as RubricPredictiveValidityReport, c as RubricRanking, r as rubricPredictiveValidity } from '../rubric-predictive-validity-C-fMteAW.js';
|
|
4
4
|
import { C as ContinuousAgreement, a as CalibrationResult, b as ContinuousCalibrationResult, c as CandidateScore, G as GoldenItem } from '../judge-calibration-7C-IDmKr.js';
|
|
5
5
|
import { S as SeriesConvergenceOptions, a as SeriesConvergenceResult } from '../series-convergence-D5OWMBg6.js';
|
|
6
|
-
import { C as CorpusAgreementReport } from '../statistics-
|
|
6
|
+
import { C as CorpusAgreementReport } from '../statistics-oUbOJe-S.js';
|
|
7
7
|
import '../store-BsVi7ncX.js';
|
|
8
8
|
import '../schema-SGWcK9wa.js';
|
|
9
9
|
import '../run-record-DksGsfgv.js';
|
package/dist/meta-eval/index.js
CHANGED
|
@@ -11,11 +11,11 @@ import {
|
|
|
11
11
|
} from "../chunk-3RF76KTD.js";
|
|
12
12
|
import {
|
|
13
13
|
rubricPredictiveValidity
|
|
14
|
-
} from "../chunk-
|
|
14
|
+
} from "../chunk-X4UCIOTZ.js";
|
|
15
15
|
import {
|
|
16
16
|
pearsonR,
|
|
17
17
|
spearmanR
|
|
18
|
-
} from "../chunk-
|
|
18
|
+
} from "../chunk-PJQFMIOX.js";
|
|
19
19
|
import {
|
|
20
20
|
aggregateLlm,
|
|
21
21
|
llmSpans
|
|
@@ -1,4 +1,4 @@
|
|
|
1
|
-
import { J as JudgeScore } from '../types-
|
|
1
|
+
import { J as JudgeScore } from '../types-CgSlO6wT.js';
|
|
2
2
|
import { AgentProfile } from '@tangle-network/agent-interface';
|
|
3
3
|
import { M as MatrixResult } from '../types-BUxNaJ8c.js';
|
|
4
4
|
import '../run-record-DksGsfgv.js';
|
package/dist/openapi.json
CHANGED
|
@@ -2,7 +2,7 @@
|
|
|
2
2
|
"openapi": "3.1.0",
|
|
3
3
|
"info": {
|
|
4
4
|
"title": "@tangle-network/agent-eval — wire protocol",
|
|
5
|
-
"version": "0.
|
|
5
|
+
"version": "0.114.0",
|
|
6
6
|
"description": "HTTP and stdio RPC interface to agent-eval. The TypeScript runtime is the source of truth; this spec is the contract that cross-language clients (Python, Rust, Go) generate from.\n\nWire-protocol version: 1.0.0. Bumps on breaking changes to request/response schemas.",
|
|
7
7
|
"contact": {
|
|
8
8
|
"name": "Tangle Network",
|
package/dist/pipelines/index.js
CHANGED
|
@@ -4,11 +4,11 @@ import {
|
|
|
4
4
|
classifyFailure,
|
|
5
5
|
compareToBaseline,
|
|
6
6
|
computeToolUseMetrics
|
|
7
|
-
} from "../chunk-
|
|
7
|
+
} from "../chunk-NYFUT3B3.js";
|
|
8
8
|
import {
|
|
9
9
|
interRaterReliability,
|
|
10
10
|
pearsonR
|
|
11
|
-
} from "../chunk-
|
|
11
|
+
} from "../chunk-PJQFMIOX.js";
|
|
12
12
|
import {
|
|
13
13
|
aggregateLlm,
|
|
14
14
|
argHash,
|
|
@@ -1,7 +1,7 @@
|
|
|
1
1
|
import { A as AgentEvalError } from './errors-oeQrLqXC.js';
|
|
2
2
|
import { R as RunRecord } from './run-record-DksGsfgv.js';
|
|
3
3
|
import { C as ChatClient } from './kind-factory-20hcaYpf.js';
|
|
4
|
-
import { a as JudgeDimension, S as Scenario, b as JudgeConfig } from './types-
|
|
4
|
+
import { a as JudgeDimension, S as Scenario, b as JudgeConfig } from './types-CgSlO6wT.js';
|
|
5
5
|
import { TCloud } from '@tangle-network/tcloud';
|
|
6
6
|
import { R as RawProviderSink } from './raw-provider-sink-C46HDghv.js';
|
|
7
7
|
import { D as DefaultVerdict } from './verdict-C9MlYujm.js';
|
|
@@ -1,7 +1,7 @@
|
|
|
1
|
-
import { S as Scenario, g as Gate, G as GateResult, n as GateContext, C as CampaignResult, p as Mutator, f as SurfaceProposer, M as MutableSurface, h as GateDecision } from './types-
|
|
2
|
-
import { e as RedTeamCase, D as Direction, b as RunCampaignOptions } from './gepa-
|
|
1
|
+
import { S as Scenario, g as Gate, G as GateResult, n as GateContext, C as CampaignResult, p as Mutator, f as SurfaceProposer, M as MutableSurface, h as GateDecision } from './types-CgSlO6wT.js';
|
|
2
|
+
import { e as RedTeamCase, D as Direction, b as RunCampaignOptions } from './gepa-DolL_Fko.js';
|
|
3
3
|
import { R as RunRecord } from './run-record-DksGsfgv.js';
|
|
4
|
-
import { a as PairedBootstrapResult } from './statistics-
|
|
4
|
+
import { a as PairedBootstrapResult } from './statistics-oUbOJe-S.js';
|
|
5
5
|
import { HostedClient, TraceSpanEvent } from './hosted/index.js';
|
|
6
6
|
import { C as CampaignStorage } from './storage-Dw_f7WMt.js';
|
|
7
7
|
|
|
@@ -375,12 +375,6 @@ declare function evolutionaryProposer<TFindings = unknown>(opts: EvolutionaryPro
|
|
|
375
375
|
* could drift from what the gate actually saw.
|
|
376
376
|
*/
|
|
377
377
|
|
|
378
|
-
/** Stable sha256 (full hex) of a surface's effective text. Code surfaces hash
|
|
379
|
-
* their worktree+base identity since the content lives in git. Distinct from
|
|
380
|
-
* `surfaceHash` (16-char content fingerprint used as a loop identity key);
|
|
381
|
-
* this is the byte-identical-verifiable content hash the provenance record +
|
|
382
|
-
* `RunRecord.promptHash` carry. */
|
|
383
|
-
declare function surfaceContentHash(surface: MutableSurface): string;
|
|
384
378
|
interface LoopProvenanceCandidate {
|
|
385
379
|
/** Generation index this candidate was proposed in. */
|
|
386
380
|
generation: number;
|
|
@@ -415,7 +409,7 @@ interface LoopProvenanceBackend {
|
|
|
415
409
|
* the bare hosted event) + backend provenance.
|
|
416
410
|
*/
|
|
417
411
|
interface LoopProvenanceRecord {
|
|
418
|
-
schema: 'tangle.loop-provenance.
|
|
412
|
+
schema: 'tangle.loop-provenance.v2';
|
|
419
413
|
runId: string;
|
|
420
414
|
runDir: string;
|
|
421
415
|
timestamp: string;
|
|
@@ -532,4 +526,4 @@ interface EmitLoopProvenanceArgs<TArtifact, TScenario extends Scenario> extends
|
|
|
532
526
|
*/
|
|
533
527
|
declare function emitLoopProvenance<TArtifact, TScenario extends Scenario>(args: EmitLoopProvenanceArgs<TArtifact, TScenario>): Promise<EmitLoopProvenanceResult>;
|
|
534
528
|
|
|
535
|
-
export { type AxisEvidence as A, type BuildEvidenceVectorOptions as B, type DefaultProductionGateOptions as D, type EvidenceVector as E, type HeldOutGateOptions as H, type LoopProvenanceRecord as L, type ObjectiveSource as O, type PowerPreflight as P, type RunEvalOptions as R, type AxisVerdict as a, type EvolutionaryProposerOptions as b, type ParetoSignificanceGateOptions as c, type PromotionObjective as d, type PromotionPolicy as e, buildEvidenceVector as f, composeGate as g, defaultProductionGate as h, evolutionaryProposer as i, heldOutGate as j, paretoSignificanceGate as k, type BuildLoopProvenanceArgs as l, type EmitLoopProvenanceArgs as m, type EmitLoopProvenanceResult as n, type LoopProvenanceBackend as o, paretoPolicy as p, type LoopProvenanceCandidate as q, runEval as r, type PowerPreflightOptions as s, buildLoopProvenanceRecord as t, emitLoopProvenance as u, loopProvenanceSpans as v, powerPreflight as w, provenanceRecordPath as x, provenanceSpansPath as y
|
|
529
|
+
export { type AxisEvidence as A, type BuildEvidenceVectorOptions as B, type DefaultProductionGateOptions as D, type EvidenceVector as E, type HeldOutGateOptions as H, type LoopProvenanceRecord as L, type ObjectiveSource as O, type PowerPreflight as P, type RunEvalOptions as R, type AxisVerdict as a, type EvolutionaryProposerOptions as b, type ParetoSignificanceGateOptions as c, type PromotionObjective as d, type PromotionPolicy as e, buildEvidenceVector as f, composeGate as g, defaultProductionGate as h, evolutionaryProposer as i, heldOutGate as j, paretoSignificanceGate as k, type BuildLoopProvenanceArgs as l, type EmitLoopProvenanceArgs as m, type EmitLoopProvenanceResult as n, type LoopProvenanceBackend as o, paretoPolicy as p, type LoopProvenanceCandidate as q, runEval as r, type PowerPreflightOptions as s, buildLoopProvenanceRecord as t, emitLoopProvenance as u, loopProvenanceSpans as v, powerPreflight as w, provenanceRecordPath as x, provenanceSpansPath as y };
|
package/dist/reporting.d.ts
CHANGED
|
@@ -1,7 +1,7 @@
|
|
|
1
1
|
export { R as RubricOutcomePair, a as RubricPredictiveValidityInput, b as RubricPredictiveValidityReport, c as RubricRanking, r as rubricPredictiveValidity } from './rubric-predictive-validity-C-fMteAW.js';
|
|
2
2
|
export { B as BootstrapOptions, a as BootstrapResult, J as JudgeReplayGateArgs, R as ReleaseConfidenceAxis, b as ReleaseConfidenceAxisName, c as ReleaseConfidenceInput, d as ReleaseConfidenceIssue, e as ReleaseConfidenceMetrics, f as ReleaseConfidenceScorecard, g as ReleaseConfidenceStatus, h as ReleaseConfidenceThresholds, i as ReleaseTraceEvidence, j as RenderReleaseReportOptions, V as Verdict, k as assertReleaseConfidence, l as bootstrapCi, m as evaluateReleaseConfidence, n as judgeReplayGate, r as renderReleaseReport } from './release-report-oBfOz8ku.js';
|
|
3
3
|
export { I as InterimReleaseConfidence, a as InterimReleaseConfidenceInput, P as PairedEvalueOptions, b as PairedEvalueSequence, c as PairedEvalueStep, S as SequentialDecision, e as evaluateInterimReleaseConfidence, p as pairedEvalueSequence } from './sequential-5iSVfzl2.js';
|
|
4
|
-
export { P as PairedBootstrapOptions, a as PairedBootstrapResult, b as benjaminiHochberg, p as pairedBootstrap, w as wilcoxonSignedRank } from './statistics-
|
|
4
|
+
export { P as PairedBootstrapOptions, a as PairedBootstrapResult, b as benjaminiHochberg, p as pairedBootstrap, w as wilcoxonSignedRank } from './statistics-oUbOJe-S.js';
|
|
5
5
|
export { G as GainDistributionBin, a as GainDistributionFigureSpec, b as GainDistributionOptions, P as ParetoFigureSpec, c as ParetoPoint, R as RESEARCH_REPORT_HARD_PAIR_FLOOR, d as ResearchReport, e as ResearchReportCandidate, f as ResearchReportDecision, g as ResearchReportMethodology, h as ResearchReportOptions, i as ResearchReportRecommendation, S as SummaryTable, j as SummaryTableOptions, k as SummaryTableRow, l as gainHistogram, p as paretoChart, r as researchReport, s as summaryTable } from './summary-report-Bz-0-t8v.js';
|
|
6
6
|
import './run-record-DksGsfgv.js';
|
|
7
7
|
import '@tangle-network/agent-interface';
|
package/dist/reporting.js
CHANGED
|
@@ -4,10 +4,10 @@ import {
|
|
|
4
4
|
evaluateReleaseConfidence,
|
|
5
5
|
judgeReplayGate,
|
|
6
6
|
renderReleaseReport
|
|
7
|
-
} from "./chunk-
|
|
7
|
+
} from "./chunk-MOXWMGPC.js";
|
|
8
8
|
import {
|
|
9
9
|
rubricPredictiveValidity
|
|
10
|
-
} from "./chunk-
|
|
10
|
+
} from "./chunk-X4UCIOTZ.js";
|
|
11
11
|
import {
|
|
12
12
|
evaluateInterimReleaseConfidence,
|
|
13
13
|
pairedEvalueSequence
|
|
@@ -18,12 +18,12 @@ import {
|
|
|
18
18
|
paretoChart,
|
|
19
19
|
researchReport,
|
|
20
20
|
summaryTable
|
|
21
|
-
} from "./chunk-
|
|
21
|
+
} from "./chunk-DPZAEKA6.js";
|
|
22
22
|
import {
|
|
23
23
|
benjaminiHochberg,
|
|
24
24
|
pairedBootstrap,
|
|
25
25
|
wilcoxonSignedRank
|
|
26
|
-
} from "./chunk-
|
|
26
|
+
} from "./chunk-PJQFMIOX.js";
|
|
27
27
|
import "./chunk-VSMTAMNK.js";
|
|
28
28
|
import "./chunk-ONWEPEDO.js";
|
|
29
29
|
import "./chunk-PZ5AY32C.js";
|
package/dist/rl.d.ts
CHANGED
|
@@ -10,7 +10,7 @@ import { R as Researcher, F as FailureMode, S as SteeringChange, E as Experiment
|
|
|
10
10
|
export { r as runEvalCampaign } from './researcher-CaH0CwFC.js';
|
|
11
11
|
import { a as VerificationReport } from './multi-layer-verifier-BsqKuLyN.js';
|
|
12
12
|
import { I as InterimReleaseConfidence } from './sequential-5iSVfzl2.js';
|
|
13
|
-
import { C as CampaignResult } from './types-
|
|
13
|
+
import { C as CampaignResult } from './types-CgSlO6wT.js';
|
|
14
14
|
import '@tangle-network/agent-interface';
|
|
15
15
|
import './errors-oeQrLqXC.js';
|
|
16
16
|
import './llm-client-DyqEH4jH.js';
|
package/dist/rl.js
CHANGED
|
@@ -10,24 +10,24 @@ import {
|
|
|
10
10
|
} from "./chunk-3RF76KTD.js";
|
|
11
11
|
import {
|
|
12
12
|
runEvalCampaign
|
|
13
|
-
} from "./chunk-
|
|
13
|
+
} from "./chunk-LOBMT6SB.js";
|
|
14
14
|
import {
|
|
15
15
|
detectRewardHacking,
|
|
16
16
|
extractVerifiableReward,
|
|
17
17
|
extractVerifiableRewardsFromRecords,
|
|
18
18
|
filterDeterministicallyRewarded
|
|
19
|
-
} from "./chunk-
|
|
19
|
+
} from "./chunk-ARU2PZFM.js";
|
|
20
20
|
import {
|
|
21
21
|
rubricPredictiveValidity
|
|
22
|
-
} from "./chunk-
|
|
22
|
+
} from "./chunk-X4UCIOTZ.js";
|
|
23
23
|
import {
|
|
24
24
|
evaluateInterimReleaseConfidence
|
|
25
25
|
} from "./chunk-MAZ26DC7.js";
|
|
26
|
-
import "./chunk-
|
|
26
|
+
import "./chunk-DPZAEKA6.js";
|
|
27
27
|
import {
|
|
28
28
|
benjaminiHochberg,
|
|
29
29
|
wilcoxonSignedRank
|
|
30
|
-
} from "./chunk-
|
|
30
|
+
} from "./chunk-PJQFMIOX.js";
|
|
31
31
|
import {
|
|
32
32
|
observationsFromRunRecords,
|
|
33
33
|
thompsonCurriculum,
|
|
@@ -1,12 +1,12 @@
|
|
|
1
1
|
import {
|
|
2
2
|
planCampaignRun,
|
|
3
3
|
runCampaign
|
|
4
|
-
} from "./chunk-
|
|
5
|
-
import "./chunk-
|
|
4
|
+
} from "./chunk-FAOEFFRT.js";
|
|
5
|
+
import "./chunk-PJQFMIOX.js";
|
|
6
6
|
import "./chunk-ONWEPEDO.js";
|
|
7
7
|
import "./chunk-PZ5AY32C.js";
|
|
8
8
|
export {
|
|
9
9
|
planCampaignRun,
|
|
10
10
|
runCampaign
|
|
11
11
|
};
|
|
12
|
-
//# sourceMappingURL=run-campaign-
|
|
12
|
+
//# sourceMappingURL=run-campaign-UADIM77S.js.map
|
|
@@ -306,6 +306,39 @@ interface PairedBootstrapOptions {
|
|
|
306
306
|
* gain is real at the confidence level. Throws on unequal sample sizes.
|
|
307
307
|
*/
|
|
308
308
|
declare function pairedBootstrap(before: number[], after: number[], opts?: PairedBootstrapOptions): PairedBootstrapResult;
|
|
309
|
+
/** Pre-registered direction for a one-sided paired sign test. */
|
|
310
|
+
type SignTestAlternative = 'greater' | 'less';
|
|
311
|
+
/** Exact one-sided sign-test result for paired numeric differences. */
|
|
312
|
+
interface PairedSignTestResult {
|
|
313
|
+
/** Total supplied differences, including zero ties. */
|
|
314
|
+
n: number;
|
|
315
|
+
/** Strictly positive differences. */
|
|
316
|
+
positive: number;
|
|
317
|
+
/** Strictly negative differences. */
|
|
318
|
+
negative: number;
|
|
319
|
+
/** Zero differences excluded from the binomial test. */
|
|
320
|
+
ties: number;
|
|
321
|
+
/** Non-zero differences used by the binomial test. */
|
|
322
|
+
nNonTies: number;
|
|
323
|
+
/** Direction of the pre-registered alternative hypothesis. */
|
|
324
|
+
alternative: SignTestAlternative;
|
|
325
|
+
/** Exact one-sided p-value under P(positive) = P(negative) = 0.5. */
|
|
326
|
+
pValue: number;
|
|
327
|
+
}
|
|
328
|
+
/**
|
|
329
|
+
* Exact one-sided sign test over paired differences.
|
|
330
|
+
*
|
|
331
|
+
* Pass `after[i] - before[i]` for each matched item. `alternative = 'greater'`
|
|
332
|
+
* tests whether positive signs are more likely than negative signs and returns
|
|
333
|
+
* `P(Binomial(nNonTies, 0.5) >= positive)`. `alternative = 'less'` treats
|
|
334
|
+
* negative signs as successes instead. With a continuous difference
|
|
335
|
+
* distribution this is the usual directional median test. Exact zero
|
|
336
|
+
* differences are ties and do not enter the binomial denominator. All-tie and
|
|
337
|
+
* empty inputs return p = 1. Every input difference must be finite, and the
|
|
338
|
+
* direction must be chosen explicitly so a caller cannot select it after
|
|
339
|
+
* seeing the signs.
|
|
340
|
+
*/
|
|
341
|
+
declare function pairedSignTest(differences: readonly number[], alternative: SignTestAlternative): PairedSignTestResult;
|
|
309
342
|
/** A binomial proportion estimate with a confidence interval. */
|
|
310
343
|
interface ProportionInterval {
|
|
311
344
|
/** Point estimate successes / n (0 when n = 0). */
|
|
@@ -458,4 +491,4 @@ declare function eProcess(opts?: EProcessOptions): EProcess;
|
|
|
458
491
|
* gate verdicts is non-reproducible by construction). */
|
|
459
492
|
declare function mulberry32(seed: number): () => number;
|
|
460
493
|
|
|
461
|
-
export {
|
|
494
|
+
export { mcnemarPower as A, mcnemarRequiredN as B, type CorpusAgreementReport as C, mulberry32 as D, type EProcessState as E, normalizeScores as F, pairedMde as G, pairedRiskDifference as H, pairedSignTest as I, pairedTTest as J, partialCredit as K, passAtK as L, type McNemarResult as M, pearsonR as N, ranks as O, type PairedBootstrapOptions as P, requiredSampleSize as Q, type RiskDifferenceResult as R, type SignTestAlternative as S, spearmanR as T, weightedComposite as U, weightedMean as V, type WeightedCompositeInput as W, wilson as X, type PairedBootstrapResult as a, benjaminiHochberg as b, type CliffsMagnitude as c, type CorpusAgreementOptions as d, type CorpusAgreementPerDimension as e, type CorpusScoreRecord as f, type EProcess as g, type EProcessOptions as h, type EProcessStep as i, type PairedSignTestResult as j, type ProportionInterval as k, type WeightedCompositeResult as l, bonferroni as m, cliffsDelta as n, cohensD as o, pairedBootstrap as p, confidenceInterval as q, corpusInterRaterAgreement as r, corpusInterRaterAgreementFromJudgeScores as s, eProcess as t, holm as u, interRaterReliability as v, wilcoxonSignedRank as w, interpretCliffs as x, mannWhitneyU as y, mcnemar as z };
|
|
@@ -116,20 +116,35 @@ interface JudgeScore {
|
|
|
116
116
|
/** Ensemble extras: each surviving judge's per-dimension scores. */
|
|
117
117
|
perJudge?: Record<string, Record<string, number>>;
|
|
118
118
|
}
|
|
119
|
-
/** A tier-4 code surface — a candidate change to the agent's
|
|
119
|
+
/** A tier-4 code surface — a finalized candidate change to the agent's
|
|
120
120
|
* IMPLEMENTATION, not its prompt. Produced by autoresearch (reads codebase +
|
|
121
|
-
* trace findings → opens a worktree).
|
|
122
|
-
*
|
|
123
|
-
* table in `docs/design/loop-taxonomy.md`. */
|
|
121
|
+
* trace findings → opens a worktree). `worktreeRef` locates the candidate;
|
|
122
|
+
* the exact commits, tree, and binary-patch digest identify it. See the
|
|
123
|
+
* improvement-tier table in `docs/design/loop-taxonomy.md`. */
|
|
124
124
|
interface CodeSurface {
|
|
125
|
-
kind: 'code';
|
|
126
|
-
/** Worktree path or git ref holding the candidate code change.
|
|
127
|
-
*
|
|
128
|
-
worktreeRef: string;
|
|
129
|
-
/**
|
|
130
|
-
baseRef
|
|
125
|
+
readonly kind: 'code';
|
|
126
|
+
/** Worktree path or git ref holding the candidate code change. This is a
|
|
127
|
+
* mutable locator and is deliberately excluded from content hashes. */
|
|
128
|
+
readonly worktreeRef: string;
|
|
129
|
+
/** Human-readable ref the worktree was forked from. Not identity-bearing. */
|
|
130
|
+
readonly baseRef: string;
|
|
131
|
+
/** Exact commit the candidate was forked from. */
|
|
132
|
+
readonly baseCommit: string;
|
|
133
|
+
/** Exact tree object for `baseCommit`. */
|
|
134
|
+
readonly baseTree: string;
|
|
135
|
+
/** Exact finalized candidate commit. */
|
|
136
|
+
readonly candidateCommit: string;
|
|
137
|
+
/** Exact tree object for `candidateCommit`. */
|
|
138
|
+
readonly candidateTree: string;
|
|
139
|
+
/** Identity of the exact patch artifact. The deployable candidate bundle
|
|
140
|
+
* carries the same descriptor plus its base64-encoded content. */
|
|
141
|
+
readonly patch: {
|
|
142
|
+
readonly format: 'git-diff-binary';
|
|
143
|
+
readonly sha256: `sha256:${string}`;
|
|
144
|
+
readonly byteLength: number;
|
|
145
|
+
};
|
|
131
146
|
/** Human summary of what changed — rendered into the auto-PR body. */
|
|
132
|
-
summary?: string;
|
|
147
|
+
readonly summary?: string;
|
|
133
148
|
}
|
|
134
149
|
/** The mutable surface a proposer changes. Tiers (see
|
|
135
150
|
* `docs/design/loop-taxonomy.md`):
|
|
@@ -162,7 +162,13 @@ imports agent-eval and implements it.
|
|
|
162
162
|
|
|
163
163
|
`MutableSurface` is the thing the proposer changes. It has tiers, least → most
|
|
164
164
|
invasive. `MutableSurface = string | CodeSurface` spans all of them: `string`
|
|
165
|
-
for tiers 1–2,
|
|
165
|
+
for tiers 1–2, and a finalized `CodeSurface` for tier 4. A code surface's
|
|
166
|
+
worktree path is only its locator; exact base/candidate commits, final tree,
|
|
167
|
+
and binary-patch digest are its portable identity. Call `verifyCodeSurface`
|
|
168
|
+
before executing the checkout so a moved ref or post-finalization mutation
|
|
169
|
+
fails before measurement. Verification hashes raw files and executable modes
|
|
170
|
+
without Git filters and rejects external symlinks or submodules whose bytes are
|
|
171
|
+
not represented by the candidate tree.
|
|
166
172
|
|
|
167
173
|
| Tier | Surface | Generator that changes it | Blast radius |
|
|
168
174
|
|---|---|---|---|
|
|
@@ -30,7 +30,7 @@ Grounded to `agent-eval/src/campaign/types.ts` unless noted.
|
|
|
30
30
|
|
|
31
31
|
| Term | Plain sentence | If you know DSPy/GEPA |
|
|
32
32
|
|---|---|---|
|
|
33
|
-
| **surface** | The one thing being changed this run — a prompt string, a JSON config string, or a code
|
|
33
|
+
| **surface** | The one thing being changed this run — a prompt string, a JSON config string, or a finalized code candidate (`MutableSurface = string \| CodeSurface`, `types.ts`). | The optimized artifact: a signature/predictor's instruction text, or the module config. |
|
|
34
34
|
| **proposer** | The strategy that, given the current surface + what failed, proposes the next batch of candidate surfaces to measure — it does **not** run the agent or score anything (`SurfaceProposer.propose`, `types.ts:286`). | The optimizer / teleprompter (MIPRO, BootstrapFewShot, GEPA's reflective proposer). |
|
|
35
35
|
| **candidate** | One proposed surface plus its human `label` and `rationale`, ready to be measured (`ProposedCandidate`, `types.ts:166`). | One trial instruction/program the optimizer wants to evaluate. |
|
|
36
36
|
| **generation** | One round of *propose → measure → rank → promote*; the loop runs up to `maxGenerations` of them (`GenerationRecord`, `types.ts:549`). | One GEPA iteration / optimization step. |
|
package/package.json
CHANGED
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
{
|
|
2
2
|
"name": "@tangle-network/agent-eval",
|
|
3
|
-
"version": "0.
|
|
3
|
+
"version": "0.114.0",
|
|
4
4
|
"description": "Evaluate and improve AI agents from runs, traces, judges, and feedback. Compare candidates, cluster failures, measure lift, and gate releases.",
|
|
5
5
|
"homepage": "https://github.com/tangle-network/agent-eval#readme",
|
|
6
6
|
"repository": {
|