@tangle-network/agent-eval 0.86.0 → 0.89.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/dist/adapters/http.d.ts +3 -3
- package/dist/adapters/langchain.d.ts +3 -3
- package/dist/adapters/otel.d.ts +6 -6
- package/dist/adversarial-DIVcDoI_.d.ts +88 -0
- package/dist/analyst/index.d.ts +11 -10
- package/dist/analyst/index.js +13 -8
- package/dist/analyst/index.js.map +1 -1
- package/dist/analyze-runs-DwCEkpO_.d.ts +81 -0
- package/dist/belief-state/index.d.ts +4 -4
- package/dist/belief-state/index.js +1 -1
- package/dist/benchmarks/index.d.ts +3 -3
- package/dist/campaign/index.d.ts +165 -18
- package/dist/campaign/index.js +289 -14
- package/dist/campaign/index.js.map +1 -1
- package/dist/chunk-45EEMHTC.js +35 -0
- package/dist/chunk-45EEMHTC.js.map +1 -0
- package/dist/{chunk-FZWAFVAA.js → chunk-4FBZZIYD.js} +2 -2
- package/dist/{chunk-YV7J7X5N.js → chunk-5HRORJQY.js} +22 -12
- package/dist/chunk-5HRORJQY.js.map +1 -0
- package/dist/{chunk-OTYQPHPL.js → chunk-6SOJM3VR.js} +5 -5
- package/dist/chunk-BOD4O7OF.js +40 -0
- package/dist/chunk-BOD4O7OF.js.map +1 -0
- package/dist/{chunk-Z7VFTS2J.js → chunk-CY6U5S3X.js} +2 -2
- package/dist/{chunk-VIDQF3F5.js → chunk-D3V5B42D.js} +5 -34
- package/dist/chunk-D3V5B42D.js.map +1 -0
- package/dist/{chunk-YGYXHNAQ.js → chunk-FIUKOSWI.js} +21 -8
- package/dist/chunk-FIUKOSWI.js.map +1 -0
- package/dist/{chunk-WJL2NJXN.js → chunk-GSH6QNNS.js} +2 -2
- package/dist/{chunk-RBNA5AZT.js → chunk-L3JOU6XM.js} +2 -2
- package/dist/{chunk-IDVBLYCY.js → chunk-LMZQ2Z4U.js} +56 -2
- package/dist/{chunk-IDVBLYCY.js.map → chunk-LMZQ2Z4U.js.map} +1 -1
- package/dist/{chunk-VUINJM5M.js → chunk-QAY5UIJO.js} +2 -193
- package/dist/chunk-QAY5UIJO.js.map +1 -0
- package/dist/{chunk-P2J6SOXT.js → chunk-QG2OVF2D.js} +5 -3
- package/dist/{chunk-P2J6SOXT.js.map → chunk-QG2OVF2D.js.map} +1 -1
- package/dist/chunk-REVYNR6C.js +100 -0
- package/dist/chunk-REVYNR6C.js.map +1 -0
- package/dist/{chunk-ZZ2HOPME.js → chunk-TWS7AZEY.js} +2 -2
- package/dist/chunk-UHMJT4T7.js +200 -0
- package/dist/chunk-UHMJT4T7.js.map +1 -0
- package/dist/chunk-UMMZHCPB.js +190 -0
- package/dist/chunk-UMMZHCPB.js.map +1 -0
- package/dist/chunk-VZSRQ272.js +149 -0
- package/dist/chunk-VZSRQ272.js.map +1 -0
- package/dist/{chunk-L5G7OUKD.js → chunk-XY4DDNEG.js} +8 -190
- package/dist/chunk-XY4DDNEG.js.map +1 -0
- package/dist/chunk-Y47J2LJ3.js +859 -0
- package/dist/chunk-Y47J2LJ3.js.map +1 -0
- package/dist/{chunk-BABOZOSN.js → chunk-ZFIBGEOL.js} +3 -3
- package/dist/chunk-ZFIBGEOL.js.map +1 -0
- package/dist/{code-agent-session-BRXmavYv.d.ts → code-agent-session-BO8nCnv3.d.ts} +1 -1
- package/dist/contract/index.d.ts +24 -95
- package/dist/contract/index.js +16 -755
- package/dist/contract/index.js.map +1 -1
- package/dist/{control-GeE8OhpN.d.ts → control-_Qb7skHX.d.ts} +2 -2
- package/dist/control.d.ts +5 -5
- package/dist/corpus-BoR-041R.d.ts +560 -0
- package/dist/cost-ledger-DuSqlw5B.d.ts +113 -0
- package/dist/counterfactual-Dwibr5IW.d.ts +85 -0
- package/dist/{dataset-B2kL-fSM.d.ts → dataset-BbGkaN2I.d.ts} +1 -1
- package/dist/{registry-DrEQ3Luj.d.ts → default-registry-zoGHUQEH.d.ts} +29 -2
- package/dist/diagnose.d.ts +251 -0
- package/dist/diagnose.js +381 -0
- package/dist/diagnose.js.map +1 -0
- package/dist/{errors-Dwqw-T_m.d.ts → errors-CzMUYo7b.d.ts} +1 -1
- package/dist/{feedback-trajectory-B3rErRsh.d.ts → feedback-trajectory-D9OVLrg9.d.ts} +1 -1
- package/dist/fuzz.d.ts +484 -0
- package/dist/fuzz.js +613 -0
- package/dist/fuzz.js.map +1 -0
- package/dist/governance/index.d.ts +4 -4
- package/dist/hosted/index.d.ts +6 -6
- package/dist/{index-DE3RXAXD.d.ts → index-Bx3gZ8xl.d.ts} +1 -1
- package/dist/index.d.ts +717 -455
- package/dist/index.js +1590 -793
- package/dist/index.js.map +1 -1
- package/dist/{insight-report-3ADTfClO.d.ts → insight-report-BBwvOh6x.d.ts} +2 -2
- package/dist/{integrity-CJzrpUua.d.ts → integrity-VJ9A7aST.d.ts} +1 -1
- package/dist/{judge-calibration-DilmB3Ml.d.ts → judge-calibration-0p2QcWNE.d.ts} +1 -1
- package/dist/{kind-factory-CVecZZG_.d.ts → kind-factory-5b7xXXOr.d.ts} +2 -2
- package/dist/{llm-client-CuUg2Mn3.d.ts → llm-client-BeEcAokY.d.ts} +1 -1
- package/dist/matrix/index.d.ts +2 -2
- package/dist/meta-eval/index.d.ts +177 -3
- package/dist/meta-eval/index.js +260 -1
- package/dist/meta-eval/index.js.map +1 -1
- package/dist/{multi-layer-verifier-DlWCXuxL.d.ts → multi-layer-verifier-DUZXrPDA.d.ts} +7 -1
- package/dist/multishot/index.d.ts +25 -11
- package/dist/multishot/index.js +36 -7
- package/dist/multishot/index.js.map +1 -1
- package/dist/openapi.json +1 -1
- package/dist/pipelines/index.js +2 -2
- package/dist/{agent-profile-D0PBIWlV.d.ts → pre-registration-DELOEJ8v.d.ts} +144 -4
- package/dist/{provenance-DPpNIOJD.d.ts → provenance-LnqRT0sS.d.ts} +5 -5
- package/dist/{red-team-DW9Ca_tj.d.ts → red-team-BXHil6c8.d.ts} +1 -1
- package/dist/{release-report-hlNtD12q.d.ts → release-report-euXIV_Sk.d.ts} +3 -3
- package/dist/reporting.d.ts +8 -8
- package/dist/reporting.js +3 -3
- package/dist/{researcher-BLPHBbNV.d.ts → researcher-DE6Gpnb4.d.ts} +4 -4
- package/dist/rl.d.ts +194 -656
- package/dist/rl.js +236 -154
- package/dist/rl.js.map +1 -1
- package/dist/{rubric-predictive-validity-CnEl9Jc8.d.ts → rubric-predictive-validity-Cy_W-hWZ.d.ts} +1 -1
- package/dist/{run-campaign-4Y5V5CN3.js → run-campaign-RDGAM5KJ.js} +3 -3
- package/dist/{run-improvement-loop-CNqQckTj.d.ts → run-improvement-loop-5z_l5zDz.d.ts} +2 -2
- package/dist/{run-record-De9VarXR.d.ts → run-record-e7vj1uZQ.d.ts} +1 -1
- package/dist/{runtime-trajectory-BLRiaifm.d.ts → runtime-trajectory-BDgfGZSr.d.ts} +1 -1
- package/dist/{semantic-concept-judge-DIEgr_6v.d.ts → semantic-concept-judge-Dn8Z6KEG.d.ts} +5 -31
- package/dist/series-convergence-D5OWMBg6.d.ts +33 -0
- package/dist/{statistics-CnC1FMbx.d.ts → statistics-C7PozGrZ.d.ts} +71 -2
- package/dist/{summary-report-Db0dDSWP.d.ts → summary-report-DGmUucwQ.d.ts} +1 -1
- package/dist/traces.d.ts +3 -3
- package/dist/traces.js +8 -6
- package/dist/{types-Cu3u_x59.d.ts → types-2VVIL04s.d.ts} +2 -2
- package/dist/{types-D7lLRYe9.d.ts → types-BU-7W85F.d.ts} +21 -1
- package/dist/{types-CqPax19X.d.ts → types-mn5Aqk7x.d.ts} +1 -1
- package/dist/{verdict-CeEgtjyI.d.ts → verdict-C9MlYujm.d.ts} +3 -0
- package/dist/wire/index.d.ts +3 -3
- package/dist/workflow/index.d.ts +12 -11
- package/dist/workflow/index.js +1 -1
- package/package.json +11 -1
- package/dist/chunk-BABOZOSN.js.map +0 -1
- package/dist/chunk-L5G7OUKD.js.map +0 -1
- package/dist/chunk-SHTXZ4O2.js +0 -113
- package/dist/chunk-SHTXZ4O2.js.map +0 -1
- package/dist/chunk-VIDQF3F5.js.map +0 -1
- package/dist/chunk-VUINJM5M.js.map +0 -1
- package/dist/chunk-YGYXHNAQ.js.map +0 -1
- package/dist/chunk-YV7J7X5N.js.map +0 -1
- /package/dist/{chunk-FZWAFVAA.js.map → chunk-4FBZZIYD.js.map} +0 -0
- /package/dist/{chunk-OTYQPHPL.js.map → chunk-6SOJM3VR.js.map} +0 -0
- /package/dist/{chunk-Z7VFTS2J.js.map → chunk-CY6U5S3X.js.map} +0 -0
- /package/dist/{chunk-WJL2NJXN.js.map → chunk-GSH6QNNS.js.map} +0 -0
- /package/dist/{chunk-RBNA5AZT.js.map → chunk-L3JOU6XM.js.map} +0 -0
- /package/dist/{chunk-ZZ2HOPME.js.map → chunk-TWS7AZEY.js.map} +0 -0
- /package/dist/{run-campaign-4Y5V5CN3.js.map → run-campaign-RDGAM5KJ.js.map} +0 -0
|
@@ -1,10 +1,10 @@
|
|
|
1
1
|
import {
|
|
2
2
|
runCampaign
|
|
3
|
-
} from "./chunk-
|
|
4
|
-
import "./chunk-
|
|
3
|
+
} from "./chunk-TWS7AZEY.js";
|
|
4
|
+
import "./chunk-LMZQ2Z4U.js";
|
|
5
5
|
import "./chunk-3BFEG2F6.js";
|
|
6
6
|
import "./chunk-PZ5AY32C.js";
|
|
7
7
|
export {
|
|
8
8
|
runCampaign
|
|
9
9
|
};
|
|
10
|
-
//# sourceMappingURL=run-campaign-
|
|
10
|
+
//# sourceMappingURL=run-campaign-RDGAM5KJ.js.map
|
|
@@ -1,5 +1,5 @@
|
|
|
1
|
-
import { L as LlmClientOptions } from './llm-client-
|
|
2
|
-
import { I as ImprovementDriver, S as Scenario, g as CampaignResult, k as GateResult, D as DispatchFn, a as JudgeConfig, L as LabeledScenarioStore, h as CampaignTraceWriter, m as GenerationRecord, M as MutableSurface, P as ParetoParent, G as Gate } from './types-
|
|
1
|
+
import { L as LlmClientOptions } from './llm-client-BeEcAokY.js';
|
|
2
|
+
import { I as ImprovementDriver, S as Scenario, g as CampaignResult, k as GateResult, D as DispatchFn, a as JudgeConfig, L as LabeledScenarioStore, h as CampaignTraceWriter, m as GenerationRecord, M as MutableSurface, P as ParetoParent, G as Gate } from './types-BU-7W85F.js';
|
|
3
3
|
|
|
4
4
|
/**
|
|
5
5
|
* @experimental
|
|
@@ -1,11 +1,10 @@
|
|
|
1
1
|
import { AxAIService } from '@ax-llm/ax';
|
|
2
|
-
import { c as TraceAnalystKindSpec } from './kind-factory-CVecZZG_.js';
|
|
3
|
-
import { b as AnalystRegistryOptions, a as AnalystRegistry } from './registry-DrEQ3Luj.js';
|
|
4
2
|
import { z } from 'zod';
|
|
5
|
-
import { c as AnalystFinding, A as Analyst, a as AnalystContext } from './types-
|
|
3
|
+
import { c as AnalystFinding, A as Analyst, a as AnalystContext } from './types-2VVIL04s.js';
|
|
4
|
+
import { T as TraceAnalystKindSpec } from './kind-factory-5b7xXXOr.js';
|
|
6
5
|
import { a as TraceAnalystSpan } from './store-C1YxJDEK.js';
|
|
7
|
-
import { L as LlmClientOptions } from './llm-client-
|
|
8
|
-
import { S as Severity } from './multi-layer-verifier-
|
|
6
|
+
import { L as LlmClientOptions } from './llm-client-BeEcAokY.js';
|
|
7
|
+
import { S as Severity } from './multi-layer-verifier-DUZXrPDA.js';
|
|
9
8
|
|
|
10
9
|
interface CreateAnalystAiConfig {
|
|
11
10
|
/** OpenAI-compatible API key forwarded as `Authorization: Bearer`.
|
|
@@ -33,31 +32,6 @@ interface CreateAnalystAiConfig {
|
|
|
33
32
|
*/
|
|
34
33
|
declare function createAnalystAi(config: CreateAnalystAiConfig): AxAIService;
|
|
35
34
|
|
|
36
|
-
/**
|
|
37
|
-
* `buildDefaultAnalystRegistry` — the canonical analyst suite, so consumers
|
|
38
|
-
* stop hand-wiring `new AnalystRegistry()` + per-kind `createTraceAnalystKind`.
|
|
39
|
-
*
|
|
40
|
-
* The deterministic `behavioralAnalyst` is ALWAYS registered (it needs no
|
|
41
|
-
* model and is model-agnostic by construction). The agentic RLM kinds are
|
|
42
|
-
* registered only when an `ai` service is supplied — so a caller with no LLM
|
|
43
|
-
* still gets the full behavioral/efficiency diagnosis, and the substrate's
|
|
44
|
-
* "any model (including no model)" guarantee holds at the suite level.
|
|
45
|
-
*/
|
|
46
|
-
|
|
47
|
-
interface DefaultAnalystRegistryOptions {
|
|
48
|
-
/** Ax service for the agentic RLM kinds. Omit → only the deterministic analyst. */
|
|
49
|
-
ai?: AxAIService;
|
|
50
|
-
/** Model for the agentic kinds (falls back to the ai service default). */
|
|
51
|
-
model?: string;
|
|
52
|
-
/** Which agentic kinds to register when `ai` is present. Default = the shipped suite. */
|
|
53
|
-
kinds?: readonly TraceAnalystKindSpec[];
|
|
54
|
-
/** Set false to omit the deterministic behavioral analyst (default: include). */
|
|
55
|
-
includeBehavioral?: boolean;
|
|
56
|
-
/** Forwarded to the AnalystRegistry constructor (signal, tags, priorFindings). */
|
|
57
|
-
registry?: AnalystRegistryOptions;
|
|
58
|
-
}
|
|
59
|
-
declare function buildDefaultAnalystRegistry(opts?: DefaultAnalystRegistryOptions): AnalystRegistry;
|
|
60
|
-
|
|
61
35
|
/**
|
|
62
36
|
* Typed `FindingSubject` — the canonical grammar every analyst kind emits.
|
|
63
37
|
*
|
|
@@ -647,4 +621,4 @@ declare function runSemanticConceptJudge(input: SemanticConceptJudgeInput, optio
|
|
|
647
621
|
*/
|
|
648
622
|
declare function createSemanticConceptJudge(options?: SemanticConceptJudgeOptions): (input: SemanticConceptJudgeInput) => Promise<SemanticConceptJudgeResult>;
|
|
649
623
|
|
|
650
|
-
export { type
|
|
624
|
+
export { type ConceptWeightStrategy as A, type BehavioralMetrics as B, type CreateAnalystAiConfig as C, DEFAULT_TRACE_ANALYST_KINDS as D, DEFAULT_COMPLEXITY_WEIGHTS as E, FAILURE_MODE_KIND_SPEC as F, SEMANTIC_CONCEPT_JUDGE_VERSION as G, type SemanticConceptJudgeResult as H, IMPROVEMENT_KIND_SPEC as I, type SuboptimalCode as J, KIND_EXPECTED_SUBJECTS as K, type SuboptimalSignal as L, computeTraceMetrics as M, createSemanticConceptJudge as N, runSemanticConceptJudge as O, type PersistedFinding as P, type SemanticConceptJudgeOptions as S, type SemanticConceptJudgeInput as a, type DiffPolicy as b, FINDING_SUBJECT_GRAMMAR_PROMPT as c, FINDING_SUBJECT_KINDS as d, type FindingSubject as e, type FindingSubjectKind as f, FindingSubjectStringSchema as g, type FindingsDiff as h, FindingsStore as i, KNOWLEDGE_GAP_KIND_SPEC as j, KNOWLEDGE_POISONING_KIND_SPEC as k, SKILL_USAGE_ANALYST as l, SkillUsageAnalyst as m, type SkillUsageRecord as n, type SkillUsageReport as o, type SkillUsageScanConfig as p, buildSkillUsageReport as q, createAnalystAi as r, defaultIsMaterial as s, diffFindings as t, emitSkillUsageFindings as u, parseFindingSubject as v, renderFindingSubject as w, type ConceptComplexity as x, type ConceptFinding as y, type ConceptSpec as z };
|
|
@@ -0,0 +1,33 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* Series convergence — detects whether a sequence of scalar measurements
|
|
3
|
+
* is stabilizing, drifting, or noisy.
|
|
4
|
+
*
|
|
5
|
+
* Lifted from ADC convergence.ts. The per-turn `ConvergenceTracker` is
|
|
6
|
+
* about progress *within* a single run; this module is about drift
|
|
7
|
+
* *across* runs (e.g. "are my nightly eval scores stabilizing?").
|
|
8
|
+
*
|
|
9
|
+
* Three signals:
|
|
10
|
+
* - stabilized: last K values have low variance (< epsilon) — done
|
|
11
|
+
* - drifting: recent trend is monotonic and beyond noise — regressing or improving
|
|
12
|
+
* - noisy: neither — keep iterating, but flag as untrustworthy for gating
|
|
13
|
+
*/
|
|
14
|
+
interface SeriesConvergenceOptions {
|
|
15
|
+
/** Window size for "recent" analysis (default 5). */
|
|
16
|
+
window?: number;
|
|
17
|
+
/** Coefficient-of-variation threshold below which the window is stabilized (default 0.05 = 5%). */
|
|
18
|
+
stableCv?: number;
|
|
19
|
+
/** Minimum monotone run length to call drift (default 3). */
|
|
20
|
+
driftRun?: number;
|
|
21
|
+
}
|
|
22
|
+
interface SeriesConvergenceResult {
|
|
23
|
+
state: 'stabilized' | 'drifting-up' | 'drifting-down' | 'noisy' | 'insufficient-data';
|
|
24
|
+
windowMean: number;
|
|
25
|
+
windowCv: number;
|
|
26
|
+
/** Longest monotonic run at the tail of the series (positive for up, negative for down). */
|
|
27
|
+
tailRun: number;
|
|
28
|
+
/** True when n ≥ window AND windowCv ≤ stableCv. */
|
|
29
|
+
stable: boolean;
|
|
30
|
+
}
|
|
31
|
+
declare function analyzeSeries(values: number[], options?: SeriesConvergenceOptions): SeriesConvergenceResult;
|
|
32
|
+
|
|
33
|
+
export { type SeriesConvergenceOptions as S, type SeriesConvergenceResult as a, analyzeSeries as b };
|
|
@@ -1,4 +1,4 @@
|
|
|
1
|
-
import {
|
|
1
|
+
import { d as ContinuousAgreementOptions, C as ContinuousAgreement } from './judge-calibration-0p2QcWNE.js';
|
|
2
2
|
import { J as JudgeScore } from './types-Croy5h7V.js';
|
|
3
3
|
|
|
4
4
|
/** Identity: dimensions already follow "higher = better" by prompt convention
|
|
@@ -245,5 +245,74 @@ interface PairedBootstrapOptions {
|
|
|
245
245
|
* gain is real at the confidence level. Throws on unequal sample sizes.
|
|
246
246
|
*/
|
|
247
247
|
declare function pairedBootstrap(before: number[], after: number[], opts?: PairedBootstrapOptions): PairedBootstrapResult;
|
|
248
|
+
interface EProcessOptions {
|
|
249
|
+
/** Type-I error budget. The process decides when wealth ≥ 1/alpha
|
|
250
|
+
* (Ville's inequality). Default 0.05. */
|
|
251
|
+
alpha?: number;
|
|
252
|
+
/** Truncation bound on the predictable bet λ ∈ [0, maxBet]. Must satisfy
|
|
253
|
+
* maxBet < 1/nullMean so every wealth factor stays strictly positive.
|
|
254
|
+
* Default 0.5. */
|
|
255
|
+
maxBet?: number;
|
|
256
|
+
/** The null boundary m₀ for H0: E[x] ≤ m₀ on x ∈ [0,1]. Default 0.5
|
|
257
|
+
* (the paired-delta encoding x = (d+1)/2 maps "no effect" to 1/2).
|
|
258
|
+
* A pre-registered minEffect shifts this — see `sequentialPairedGate`. */
|
|
259
|
+
nullMean?: number;
|
|
260
|
+
}
|
|
261
|
+
interface EProcessStep {
|
|
262
|
+
/** Current wealth W_n — the e-value against H0 after n observations. */
|
|
263
|
+
wealth: number;
|
|
264
|
+
/** Observations consumed so far. */
|
|
265
|
+
n: number;
|
|
266
|
+
/** True from the first n where W_n ≥ 1/alpha onward (sticky). */
|
|
267
|
+
decided: boolean;
|
|
268
|
+
}
|
|
269
|
+
interface EProcessState extends EProcessStep {
|
|
270
|
+
alpha: number;
|
|
271
|
+
maxBet: number;
|
|
272
|
+
nullMean: number;
|
|
273
|
+
/** The decision boundary 1/alpha. */
|
|
274
|
+
threshold: number;
|
|
275
|
+
/** Observation count at the first threshold crossing; undefined until decided. */
|
|
276
|
+
decidedAtN?: number;
|
|
277
|
+
}
|
|
278
|
+
interface EProcess {
|
|
279
|
+
/** Consume one observation x ∈ [0,1]. Throws on non-finite / out-of-range
|
|
280
|
+
* input — a silent clamp would corrupt the type-I guarantee. */
|
|
281
|
+
update(x: number): EProcessStep;
|
|
282
|
+
state(): EProcessState;
|
|
283
|
+
}
|
|
284
|
+
/**
|
|
285
|
+
* Betting test-martingale for bounded observations — the e-process core of
|
|
286
|
+
* anytime-valid sequential testing (Waudby-Smith & Ramdas, "Estimating means
|
|
287
|
+
* of bounded random variables by betting", JRSS-B 2024).
|
|
288
|
+
*
|
|
289
|
+
* Observations x_i ∈ [0,1]; H0: E[x] ≤ m₀ (`nullMean`, default 1/2). Wealth
|
|
290
|
+
*
|
|
291
|
+
* W_t = Π_{i≤t} (1 + λ_i (x_i − m₀)), W_0 = 1
|
|
292
|
+
*
|
|
293
|
+
* with the truncated GROW-style plug-in bet computed from PRIOR observations:
|
|
294
|
+
*
|
|
295
|
+
* λ_i = clamp((μ̂_{i−1} − m₀) / (σ̂²_{i−1} + (μ̂_{i−1} − m₀)²), 0, maxBet)
|
|
296
|
+
*
|
|
297
|
+
* where μ̂/σ̂² are the shrunk running estimates μ̂_t = (1/2 + Σx_i)/(t+1),
|
|
298
|
+
* σ̂²_t = (1/4 + Σ(x_i − μ̂_i)²)/(t+1).
|
|
299
|
+
*
|
|
300
|
+
* PREDICTABILITY INVARIANT (load-bearing): λ_i is a function of x_1..x_{i−1}
|
|
301
|
+
* ONLY — it may never see x_i. With λ_i ≥ 0 predictable, each factor has
|
|
302
|
+
* E[1 + λ_i(x_i − m₀) | past] ≤ 1 under H0, so W is a nonnegative
|
|
303
|
+
* supermartingale and Ville's inequality gives P(∃t: W_t ≥ 1/α) ≤ α — the
|
|
304
|
+
* type-I guarantee holds at ANY data-dependent stopping time. λ_1 is always 0
|
|
305
|
+
* (no prior evidence), so the first observation never moves wealth.
|
|
306
|
+
*
|
|
307
|
+
* `decided` latches at the first crossing W_t ≥ 1/α and never un-latches;
|
|
308
|
+
* wealth keeps updating after the crossing (the e-process remains valid), but
|
|
309
|
+
* the decision time is the first crossing.
|
|
310
|
+
*/
|
|
311
|
+
declare function eProcess(opts?: EProcessOptions): EProcess;
|
|
312
|
+
/** Tiny seedable PRNG (mulberry32) — deterministic resampling/shuffling, not
|
|
313
|
+
* cryptographic. Exported so e-process shuffles and bootstrap resampling
|
|
314
|
+
* share ONE PRNG implementation; a seed is REQUIRED (unseeded randomness in
|
|
315
|
+
* gate verdicts is non-reproducible by construction). */
|
|
316
|
+
declare function mulberry32(seed: number): () => number;
|
|
248
317
|
|
|
249
|
-
export { type
|
|
318
|
+
export { partialCredit as A, requiredSampleSize as B, type CorpusAgreementReport as C, weightedComposite as D, type EProcessState as E, weightedMean as F, type PairedBootstrapOptions as P, type WeightedCompositeInput as W, type PairedBootstrapResult as a, benjaminiHochberg as b, type CliffsMagnitude as c, type CorpusAgreementOptions as d, type CorpusAgreementPerDimension as e, type CorpusScoreRecord as f, type EProcess as g, type EProcessOptions as h, type EProcessStep as i, type WeightedCompositeResult as j, bonferroni as k, cliffsDelta as l, cohensD as m, confidenceInterval as n, corpusInterRaterAgreement as o, pairedBootstrap as p, corpusInterRaterAgreementFromJudgeScores as q, eProcess as r, interRaterReliability as s, interpretCliffs as t, mannWhitneyU as u, mulberry32 as v, wilcoxonSignedRank as w, normalizeScores as x, pairedMde as y, pairedTTest as z };
|
package/dist/traces.d.ts
CHANGED
|
@@ -1,9 +1,9 @@
|
|
|
1
|
-
import { N as NotFoundError, R as ReplayError } from './errors-
|
|
1
|
+
import { N as NotFoundError, R as ReplayError } from './errors-CzMUYo7b.js';
|
|
2
2
|
import { P as ProviderRedactor, R as RawProviderSink, d as RawProviderEvent } from './raw-provider-sink-C46HDghv.js';
|
|
3
3
|
export { F as FileSystemRawProviderSink, a as FileSystemRawProviderSinkOptions, I as InMemoryRawProviderSink, b as InMemoryRawProviderSinkOptions, N as NoopRawProviderSink, c as RawProviderDirection, e as RawProviderSinkFilter, f as defaultProviderRedactor, p as providerFromBaseUrl } from './raw-provider-sink-C46HDghv.js';
|
|
4
4
|
import { a as RunCompleteHookContext, R as RunCompleteHook } from './emitter-DEZwY14K.js';
|
|
5
5
|
export { S as SpanHandle, T as TraceEmitter, b as TraceEmitterOptions, l as llmSpanFromProvider } from './emitter-DEZwY14K.js';
|
|
6
|
-
export { b as RunIntegrityError, R as RunIntegrityExpectations, c as RunIntegrityIssue, d as RunIntegrityIssueCode, a as RunIntegrityReport, e as assertRunCaptured, t as throwIfRunIncomplete } from './integrity-
|
|
6
|
+
export { b as RunIntegrityError, R as RunIntegrityExpectations, c as RunIntegrityIssue, d as RunIntegrityIssueCode, a as RunIntegrityReport, e as assertRunCaptured, t as throwIfRunIncomplete } from './integrity-VJ9A7aST.js';
|
|
7
7
|
import { T as TraceStore } from './store-CKUAgsJz.js';
|
|
8
8
|
export { E as EventFilter, F as FileSystemTraceStore, a as FileSystemTraceStoreOptions, I as InMemoryTraceStore, R as RunFilter, S as SpanFilter } from './store-CKUAgsJz.js';
|
|
9
9
|
export { a as aggregateLlm, b as argHash, g as groupBy, j as judgeSpans, l as llmSpans, r as runFailureClass, c as runsForScenario, t as toolSpans } from './query-CqTxMwDw.js';
|
|
@@ -14,7 +14,7 @@ import { A as AnalyzeTracesOptions, b as AnalyzeTracesResult } from './analyst-C
|
|
|
14
14
|
export { a as AnalyzeTracesInput, c as AnalyzeTracesTurnSnapshot, d as analyzeTraces } from './analyst-C8HHvfJp.js';
|
|
15
15
|
import { h as TraceAnalystSpanKind, i as TraceAnalystSpanStatus, T as TraceAnalysisStore, g as TraceAnalystFilters, b as DatasetOverview, Q as QueryTracesPage, l as ViewTraceResult, V as ViewSpansResult, c as SearchTraceResult, S as SearchSpanResult } from './store-C1YxJDEK.js';
|
|
16
16
|
export { D as DEFAULT_TRACE_ANALYST_BUDGETS, E as ErrorCluster, d as SpanMatchRecord, e as TRACE_ANALYST_TRUNCATION_MARKER_PREFIX, f as TraceAnalystByteBudgets, a as TraceAnalystSpan, j as TraceAnalystTraceSummary, k as ViewTraceOversized } from './store-C1YxJDEK.js';
|
|
17
|
-
import { a as RunSplitTag, b as RunTokenUsage, R as RunRecord } from './run-record-
|
|
17
|
+
import { a as RunSplitTag, b as RunTokenUsage, R as RunRecord } from './run-record-e7vj1uZQ.js';
|
|
18
18
|
import { AxFunction } from '@ax-llm/ax';
|
|
19
19
|
|
|
20
20
|
/**
|
package/dist/traces.js
CHANGED
|
@@ -28,7 +28,7 @@ import {
|
|
|
28
28
|
scoreTraceInsightReadiness,
|
|
29
29
|
tokenizeDomainWords,
|
|
30
30
|
traceAnalystOnRunComplete
|
|
31
|
-
} from "./chunk-
|
|
31
|
+
} from "./chunk-QG2OVF2D.js";
|
|
32
32
|
import {
|
|
33
33
|
DEFAULT_REDACTION_RULES,
|
|
34
34
|
REDACTION_VERSION,
|
|
@@ -55,16 +55,18 @@ import {
|
|
|
55
55
|
isToolSpan
|
|
56
56
|
} from "./chunk-5BKGXME7.js";
|
|
57
57
|
import {
|
|
58
|
-
DEFAULT_TRACE_ANALYST_BUDGETS,
|
|
59
|
-
OtlpFileTraceStore,
|
|
60
|
-
SpanNotFoundError,
|
|
61
58
|
TRACE_ANALYST_ACTOR_DESCRIPTION,
|
|
62
59
|
TRACE_ANALYST_ACTOR_DESCRIPTION_VERSION,
|
|
63
60
|
TRACE_ANALYST_SUBAGENT_DESCRIPTION,
|
|
61
|
+
analyzeTraces
|
|
62
|
+
} from "./chunk-UHMJT4T7.js";
|
|
63
|
+
import {
|
|
64
|
+
DEFAULT_TRACE_ANALYST_BUDGETS,
|
|
65
|
+
OtlpFileTraceStore,
|
|
66
|
+
SpanNotFoundError,
|
|
64
67
|
TRACE_ANALYST_TRUNCATION_MARKER_PREFIX,
|
|
65
68
|
TraceFileMissingError,
|
|
66
69
|
TraceNotFoundError,
|
|
67
|
-
analyzeTraces,
|
|
68
70
|
asNumber,
|
|
69
71
|
asString,
|
|
70
72
|
buildTraceAnalystTools,
|
|
@@ -76,7 +78,7 @@ import {
|
|
|
76
78
|
readOtlpStatus,
|
|
77
79
|
stringField,
|
|
78
80
|
traceAnalystFunctionGroup
|
|
79
|
-
} from "./chunk-
|
|
81
|
+
} from "./chunk-QAY5UIJO.js";
|
|
80
82
|
import {
|
|
81
83
|
RunIntegrityError,
|
|
82
84
|
assertRunCaptured,
|
|
@@ -1,7 +1,7 @@
|
|
|
1
|
-
import { R as RunRecord } from './run-record-
|
|
1
|
+
import { R as RunRecord } from './run-record-e7vj1uZQ.js';
|
|
2
2
|
import { T as TraceAnalysisStore } from './store-C1YxJDEK.js';
|
|
3
3
|
import { a as JudgeInput } from './types-Croy5h7V.js';
|
|
4
|
-
import { b as LlmCallRequest, c as LlmCallResult } from './llm-client-
|
|
4
|
+
import { b as LlmCallRequest, c as LlmCallResult } from './llm-client-BeEcAokY.js';
|
|
5
5
|
|
|
6
6
|
/**
|
|
7
7
|
* ChatClient — the single LLM abstraction analysts call.
|
|
@@ -1,4 +1,4 @@
|
|
|
1
|
-
import { b as RunTokenUsage } from './run-record-
|
|
1
|
+
import { b as RunTokenUsage } from './run-record-e7vj1uZQ.js';
|
|
2
2
|
|
|
3
3
|
/**
|
|
4
4
|
* @experimental
|
|
@@ -93,10 +93,30 @@ interface JudgeConfig<TArtifact, TScenario extends Scenario = Scenario> {
|
|
|
93
93
|
}): JudgeScore | Promise<JudgeScore>;
|
|
94
94
|
appliesTo?: (scenario: TScenario) => boolean;
|
|
95
95
|
}
|
|
96
|
+
/** The canonical judge verdict shape — one declaration, shared by campaign
|
|
97
|
+
* judges and the multishot judge runner (which re-exports this type).
|
|
98
|
+
*
|
|
99
|
+
* Scale is PRODUCER-DEFINED: campaign convention is [0,1]; the legacy
|
|
100
|
+
* multishot runner emits 0-10. Cross-scale comparison must go through
|
|
101
|
+
* `detectScale` (src/campaign/gates/statistical-heldout.ts, used by
|
|
102
|
+
* promotion-policy) — never renormalize a producer's values in place, as
|
|
103
|
+
* downstream thresholds (`composite >= 5` in multishot/matrix.ts, live-soak
|
|
104
|
+
* `>= 7` gates) key on the producer's native scale. */
|
|
96
105
|
interface JudgeScore {
|
|
97
106
|
dimensions: Record<string, number>;
|
|
98
107
|
composite: number;
|
|
99
108
|
notes: string;
|
|
109
|
+
/** Set when the judge itself failed (call error, unparseable output).
|
|
110
|
+
* `composite`/`dimensions` carry no signal — aggregators MUST exclude
|
|
111
|
+
* failed scores from means instead of folding them into zeros. */
|
|
112
|
+
failed?: true;
|
|
113
|
+
/** Ensemble extras (populated by `ensembleJudge`): max per-dimension
|
|
114
|
+
* spread across surviving judges — the inter-rater signal. */
|
|
115
|
+
maxDisagreement?: number;
|
|
116
|
+
/** Ensemble extras: judge identities whose verdict failed. */
|
|
117
|
+
failedJudges?: string[];
|
|
118
|
+
/** Ensemble extras: each surviving judge's per-dimension scores. */
|
|
119
|
+
perJudge?: Record<string, Record<string, number>>;
|
|
100
120
|
}
|
|
101
121
|
/** @experimental A tier-4 code surface — a candidate change to the agent's
|
|
102
122
|
* IMPLEMENTATION, not its prompt. Produced by autoresearch (reads codebase +
|
|
@@ -17,6 +17,9 @@
|
|
|
17
17
|
* Minimal verdict shape — `valid` + `score` are required; `scores` +
|
|
18
18
|
* `notes` are optional surface. Validators that need richer shapes
|
|
19
19
|
* parameterise `Validator<Output, MyVerdict>` with their own type.
|
|
20
|
+
*
|
|
21
|
+
* Need structured extras? Extend DefaultVerdict with typed fields — never
|
|
22
|
+
* serialize extras into `notes`.
|
|
20
23
|
*/
|
|
21
24
|
interface DefaultVerdict {
|
|
22
25
|
/** Whether the output meets the validator's pass criteria. */
|
package/dist/wire/index.d.ts
CHANGED
|
@@ -1,4 +1,4 @@
|
|
|
1
|
-
import { F as FeedbackTrajectoryStore } from '../feedback-trajectory-
|
|
1
|
+
import { F as FeedbackTrajectoryStore } from '../feedback-trajectory-D9OVLrg9.js';
|
|
2
2
|
import { T as TraceStore } from '../store-CKUAgsJz.js';
|
|
3
3
|
import { z } from 'zod';
|
|
4
4
|
import { OpenAPIObject } from 'openapi3-ts/oas31';
|
|
@@ -8,8 +8,8 @@ import { Hono } from 'hono';
|
|
|
8
8
|
import '../control-runtime-DuFBYg7A.js';
|
|
9
9
|
import '../emitter-DEZwY14K.js';
|
|
10
10
|
import '../schema-m0gsnbt3.js';
|
|
11
|
-
import '../dataset-
|
|
12
|
-
import '../errors-
|
|
11
|
+
import '../dataset-BbGkaN2I.js';
|
|
12
|
+
import '../errors-CzMUYo7b.js';
|
|
13
13
|
|
|
14
14
|
declare const RubricDimensionSchema: z.ZodObject<{
|
|
15
15
|
id: z.ZodString;
|
package/dist/workflow/index.d.ts
CHANGED
|
@@ -1,25 +1,26 @@
|
|
|
1
1
|
import { W as WorkflowTopology } from '../harness-optimizer-EnEnQPsr.js';
|
|
2
|
-
import { a as RunSplitTag, b as RunTokenUsage, R as RunRecord } from '../run-record-
|
|
3
|
-
import { c as AnalystFinding, h as AnalystSeverity, E as EvidenceRef } from '../types-
|
|
4
|
-
import { F as FailureClusterInsight } from '../insight-report-
|
|
5
|
-
import { a as VerificationReport, L as LayerResult } from '../multi-layer-verifier-
|
|
2
|
+
import { a as RunSplitTag, b as RunTokenUsage, R as RunRecord } from '../run-record-e7vj1uZQ.js';
|
|
3
|
+
import { c as AnalystFinding, h as AnalystSeverity, E as EvidenceRef } from '../types-2VVIL04s.js';
|
|
4
|
+
import { F as FailureClusterInsight } from '../insight-report-BBwvOh6x.js';
|
|
5
|
+
import { a as VerificationReport, L as LayerResult } from '../multi-layer-verifier-DUZXrPDA.js';
|
|
6
6
|
import { F as FailureClusterReport } from '../failure-cluster-CL7IVgkJ.js';
|
|
7
7
|
import { R as RedactionRule, a as RedactionReport } from '../redact-B40YG2M_.js';
|
|
8
|
-
import { D as DatasetSplit } from '../dataset-
|
|
9
|
-
import { a as FeedbackTrajectory } from '../feedback-trajectory-
|
|
10
|
-
import { a as PairedBootstrapResult } from '../statistics-
|
|
8
|
+
import { D as DatasetSplit } from '../dataset-BbGkaN2I.js';
|
|
9
|
+
import { a as FeedbackTrajectory } from '../feedback-trajectory-D9OVLrg9.js';
|
|
10
|
+
import { a as PairedBootstrapResult } from '../statistics-C7PozGrZ.js';
|
|
11
11
|
import '../pareto-E-pembql.js';
|
|
12
12
|
import '../run-critic-BAIjX99r.js';
|
|
13
13
|
import '../schema-m0gsnbt3.js';
|
|
14
14
|
import '../store-CKUAgsJz.js';
|
|
15
|
-
import '../errors-
|
|
15
|
+
import '../errors-CzMUYo7b.js';
|
|
16
16
|
import '../store-C1YxJDEK.js';
|
|
17
17
|
import '../types-Croy5h7V.js';
|
|
18
18
|
import '@tangle-network/tcloud';
|
|
19
|
-
import '../llm-client-
|
|
19
|
+
import '../llm-client-BeEcAokY.js';
|
|
20
20
|
import '../raw-provider-sink-C46HDghv.js';
|
|
21
|
-
import '../summary-report-
|
|
22
|
-
import '../judge-calibration-
|
|
21
|
+
import '../summary-report-DGmUucwQ.js';
|
|
22
|
+
import '../judge-calibration-0p2QcWNE.js';
|
|
23
|
+
import '../verdict-C9MlYujm.js';
|
|
23
24
|
import '../control-runtime-DuFBYg7A.js';
|
|
24
25
|
import '../emitter-DEZwY14K.js';
|
|
25
26
|
|
package/dist/workflow/index.js
CHANGED
package/package.json
CHANGED
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
{
|
|
2
2
|
"name": "@tangle-network/agent-eval",
|
|
3
|
-
"version": "0.
|
|
3
|
+
"version": "0.89.0",
|
|
4
4
|
"description": "Evaluate and improve AI agents from runs, traces, judges, and feedback. Compare candidates, cluster failures, measure lift, and gate releases.",
|
|
5
5
|
"homepage": "https://github.com/tangle-network/agent-eval#readme",
|
|
6
6
|
"repository": {
|
|
@@ -39,6 +39,16 @@
|
|
|
39
39
|
"import": "./dist/rl.js",
|
|
40
40
|
"default": "./dist/rl.js"
|
|
41
41
|
},
|
|
42
|
+
"./diagnose": {
|
|
43
|
+
"types": "./dist/diagnose.d.ts",
|
|
44
|
+
"import": "./dist/diagnose.js",
|
|
45
|
+
"default": "./dist/diagnose.js"
|
|
46
|
+
},
|
|
47
|
+
"./fuzz": {
|
|
48
|
+
"types": "./dist/fuzz.d.ts",
|
|
49
|
+
"import": "./dist/fuzz.js",
|
|
50
|
+
"default": "./dist/fuzz.js"
|
|
51
|
+
},
|
|
42
52
|
"./traces": {
|
|
43
53
|
"types": "./dist/traces.d.ts",
|
|
44
54
|
"import": "./dist/traces.js",
|