@tangle-network/agent-eval 0.86.0 → 0.90.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (140) hide show
  1. package/CHANGELOG.md +9 -1
  2. package/dist/adapters/http.d.ts +3 -3
  3. package/dist/adapters/langchain.d.ts +3 -3
  4. package/dist/adapters/otel.d.ts +6 -6
  5. package/dist/adversarial-DIVcDoI_.d.ts +88 -0
  6. package/dist/analyst/index.d.ts +11 -10
  7. package/dist/analyst/index.js +13 -8
  8. package/dist/analyst/index.js.map +1 -1
  9. package/dist/analyze-runs-DwCEkpO_.d.ts +81 -0
  10. package/dist/belief-state/index.d.ts +4 -4
  11. package/dist/belief-state/index.js +1 -1
  12. package/dist/benchmarks/index.d.ts +3 -3
  13. package/dist/campaign/index.d.ts +165 -18
  14. package/dist/campaign/index.js +289 -14
  15. package/dist/campaign/index.js.map +1 -1
  16. package/dist/chunk-45EEMHTC.js +35 -0
  17. package/dist/chunk-45EEMHTC.js.map +1 -0
  18. package/dist/{chunk-FZWAFVAA.js → chunk-4FBZZIYD.js} +2 -2
  19. package/dist/{chunk-YV7J7X5N.js → chunk-5HRORJQY.js} +22 -12
  20. package/dist/chunk-5HRORJQY.js.map +1 -0
  21. package/dist/{chunk-OTYQPHPL.js → chunk-6SOJM3VR.js} +5 -5
  22. package/dist/chunk-BOD4O7OF.js +40 -0
  23. package/dist/chunk-BOD4O7OF.js.map +1 -0
  24. package/dist/{chunk-Z7VFTS2J.js → chunk-CY6U5S3X.js} +2 -2
  25. package/dist/{chunk-VIDQF3F5.js → chunk-D3V5B42D.js} +5 -34
  26. package/dist/chunk-D3V5B42D.js.map +1 -0
  27. package/dist/{chunk-YGYXHNAQ.js → chunk-FIUKOSWI.js} +21 -8
  28. package/dist/chunk-FIUKOSWI.js.map +1 -0
  29. package/dist/{chunk-WJL2NJXN.js → chunk-GSH6QNNS.js} +2 -2
  30. package/dist/{chunk-RBNA5AZT.js → chunk-L3JOU6XM.js} +2 -2
  31. package/dist/{chunk-IDVBLYCY.js → chunk-LMZQ2Z4U.js} +56 -2
  32. package/dist/{chunk-IDVBLYCY.js.map → chunk-LMZQ2Z4U.js.map} +1 -1
  33. package/dist/{chunk-VUINJM5M.js → chunk-QAY5UIJO.js} +2 -193
  34. package/dist/chunk-QAY5UIJO.js.map +1 -0
  35. package/dist/{chunk-P2J6SOXT.js → chunk-QG2OVF2D.js} +5 -3
  36. package/dist/{chunk-P2J6SOXT.js.map → chunk-QG2OVF2D.js.map} +1 -1
  37. package/dist/chunk-REVYNR6C.js +100 -0
  38. package/dist/chunk-REVYNR6C.js.map +1 -0
  39. package/dist/chunk-STGVSCDH.js +202 -0
  40. package/dist/chunk-STGVSCDH.js.map +1 -0
  41. package/dist/{chunk-ZZ2HOPME.js → chunk-TWS7AZEY.js} +2 -2
  42. package/dist/chunk-UHMJT4T7.js +200 -0
  43. package/dist/chunk-UHMJT4T7.js.map +1 -0
  44. package/dist/chunk-UMMZHCPB.js +190 -0
  45. package/dist/chunk-UMMZHCPB.js.map +1 -0
  46. package/dist/chunk-VZSRQ272.js +149 -0
  47. package/dist/chunk-VZSRQ272.js.map +1 -0
  48. package/dist/{chunk-L5G7OUKD.js → chunk-XY4DDNEG.js} +8 -190
  49. package/dist/chunk-XY4DDNEG.js.map +1 -0
  50. package/dist/chunk-Y47J2LJ3.js +859 -0
  51. package/dist/chunk-Y47J2LJ3.js.map +1 -0
  52. package/dist/{chunk-BABOZOSN.js → chunk-ZFIBGEOL.js} +3 -3
  53. package/dist/chunk-ZFIBGEOL.js.map +1 -0
  54. package/dist/{code-agent-session-BRXmavYv.d.ts → code-agent-session-BO8nCnv3.d.ts} +1 -1
  55. package/dist/contract/index.d.ts +24 -95
  56. package/dist/contract/index.js +25 -764
  57. package/dist/contract/index.js.map +1 -1
  58. package/dist/{control-GeE8OhpN.d.ts → control-_Qb7skHX.d.ts} +2 -2
  59. package/dist/control.d.ts +5 -5
  60. package/dist/corpus-BoR-041R.d.ts +560 -0
  61. package/dist/cost-ledger-DuSqlw5B.d.ts +113 -0
  62. package/dist/counterfactual-Dwibr5IW.d.ts +85 -0
  63. package/dist/{dataset-B2kL-fSM.d.ts → dataset-BbGkaN2I.d.ts} +1 -1
  64. package/dist/{registry-DrEQ3Luj.d.ts → default-registry-zoGHUQEH.d.ts} +29 -2
  65. package/dist/diagnose.d.ts +251 -0
  66. package/dist/diagnose.js +381 -0
  67. package/dist/diagnose.js.map +1 -0
  68. package/dist/{errors-Dwqw-T_m.d.ts → errors-CzMUYo7b.d.ts} +1 -1
  69. package/dist/{feedback-trajectory-B3rErRsh.d.ts → feedback-trajectory-D9OVLrg9.d.ts} +1 -1
  70. package/dist/fuzz.d.ts +484 -0
  71. package/dist/fuzz.js +613 -0
  72. package/dist/fuzz.js.map +1 -0
  73. package/dist/governance/index.d.ts +4 -4
  74. package/dist/hosted/index.d.ts +6 -6
  75. package/dist/{index-DE3RXAXD.d.ts → index-Bx3gZ8xl.d.ts} +1 -1
  76. package/dist/index.d.ts +718 -455
  77. package/dist/index.js +1604 -793
  78. package/dist/index.js.map +1 -1
  79. package/dist/{insight-report-3ADTfClO.d.ts → insight-report-BBwvOh6x.d.ts} +2 -2
  80. package/dist/{integrity-CJzrpUua.d.ts → integrity-VJ9A7aST.d.ts} +1 -1
  81. package/dist/{judge-calibration-DilmB3Ml.d.ts → judge-calibration-0p2QcWNE.d.ts} +1 -1
  82. package/dist/{kind-factory-CVecZZG_.d.ts → kind-factory-5b7xXXOr.d.ts} +2 -2
  83. package/dist/{llm-client-CuUg2Mn3.d.ts → llm-client-BeEcAokY.d.ts} +1 -1
  84. package/dist/matrix/index.d.ts +2 -2
  85. package/dist/meta-eval/index.d.ts +177 -3
  86. package/dist/meta-eval/index.js +260 -1
  87. package/dist/meta-eval/index.js.map +1 -1
  88. package/dist/{multi-layer-verifier-DlWCXuxL.d.ts → multi-layer-verifier-DUZXrPDA.d.ts} +7 -1
  89. package/dist/multishot/index.d.ts +25 -11
  90. package/dist/multishot/index.js +36 -7
  91. package/dist/multishot/index.js.map +1 -1
  92. package/dist/openapi.json +1 -1
  93. package/dist/perf/index.d.ts +123 -0
  94. package/dist/perf/index.js +18 -0
  95. package/dist/pipelines/index.js +2 -2
  96. package/dist/{agent-profile-D0PBIWlV.d.ts → pre-registration-DELOEJ8v.d.ts} +144 -4
  97. package/dist/{provenance-DPpNIOJD.d.ts → provenance-LnqRT0sS.d.ts} +5 -5
  98. package/dist/{red-team-DW9Ca_tj.d.ts → red-team-BXHil6c8.d.ts} +1 -1
  99. package/dist/{release-report-hlNtD12q.d.ts → release-report-euXIV_Sk.d.ts} +3 -3
  100. package/dist/reporting.d.ts +8 -8
  101. package/dist/reporting.js +3 -3
  102. package/dist/{researcher-BLPHBbNV.d.ts → researcher-DE6Gpnb4.d.ts} +4 -4
  103. package/dist/rl.d.ts +194 -656
  104. package/dist/rl.js +231 -149
  105. package/dist/rl.js.map +1 -1
  106. package/dist/{rubric-predictive-validity-CnEl9Jc8.d.ts → rubric-predictive-validity-Cy_W-hWZ.d.ts} +1 -1
  107. package/dist/{run-campaign-4Y5V5CN3.js → run-campaign-RDGAM5KJ.js} +3 -3
  108. package/dist/run-campaign-RDGAM5KJ.js.map +1 -0
  109. package/dist/{run-improvement-loop-CNqQckTj.d.ts → run-improvement-loop-5z_l5zDz.d.ts} +2 -2
  110. package/dist/{run-record-De9VarXR.d.ts → run-record-e7vj1uZQ.d.ts} +1 -1
  111. package/dist/{runtime-trajectory-BLRiaifm.d.ts → runtime-trajectory-BDgfGZSr.d.ts} +1 -1
  112. package/dist/{semantic-concept-judge-DIEgr_6v.d.ts → semantic-concept-judge-Dn8Z6KEG.d.ts} +5 -31
  113. package/dist/series-convergence-D5OWMBg6.d.ts +33 -0
  114. package/dist/{statistics-CnC1FMbx.d.ts → statistics-C7PozGrZ.d.ts} +71 -2
  115. package/dist/{summary-report-Db0dDSWP.d.ts → summary-report-DGmUucwQ.d.ts} +1 -1
  116. package/dist/traces.d.ts +3 -3
  117. package/dist/traces.js +8 -6
  118. package/dist/{types-Cu3u_x59.d.ts → types-2VVIL04s.d.ts} +2 -2
  119. package/dist/{types-D7lLRYe9.d.ts → types-BU-7W85F.d.ts} +21 -1
  120. package/dist/{types-CqPax19X.d.ts → types-mn5Aqk7x.d.ts} +1 -1
  121. package/dist/{verdict-CeEgtjyI.d.ts → verdict-C9MlYujm.d.ts} +3 -0
  122. package/dist/wire/index.d.ts +3 -3
  123. package/dist/workflow/index.d.ts +12 -11
  124. package/dist/workflow/index.js +1 -1
  125. package/package.json +16 -1
  126. package/dist/chunk-BABOZOSN.js.map +0 -1
  127. package/dist/chunk-L5G7OUKD.js.map +0 -1
  128. package/dist/chunk-SHTXZ4O2.js +0 -113
  129. package/dist/chunk-SHTXZ4O2.js.map +0 -1
  130. package/dist/chunk-VIDQF3F5.js.map +0 -1
  131. package/dist/chunk-VUINJM5M.js.map +0 -1
  132. package/dist/chunk-YGYXHNAQ.js.map +0 -1
  133. package/dist/chunk-YV7J7X5N.js.map +0 -1
  134. /package/dist/{chunk-FZWAFVAA.js.map → chunk-4FBZZIYD.js.map} +0 -0
  135. /package/dist/{chunk-OTYQPHPL.js.map → chunk-6SOJM3VR.js.map} +0 -0
  136. /package/dist/{chunk-Z7VFTS2J.js.map → chunk-CY6U5S3X.js.map} +0 -0
  137. /package/dist/{chunk-WJL2NJXN.js.map → chunk-GSH6QNNS.js.map} +0 -0
  138. /package/dist/{chunk-RBNA5AZT.js.map → chunk-L3JOU6XM.js.map} +0 -0
  139. /package/dist/{chunk-ZZ2HOPME.js.map → chunk-TWS7AZEY.js.map} +0 -0
  140. /package/dist/{run-campaign-4Y5V5CN3.js.map → perf/index.js.map} +0 -0
@@ -1,4 +1,4 @@
1
- import { R as RunRecord } from './run-record-De9VarXR.js';
1
+ import { R as RunRecord } from './run-record-e7vj1uZQ.js';
2
2
  import { b as OutcomeStore } from './outcome-store-rnXLEqSn.js';
3
3
 
4
4
  /**
@@ -1,10 +1,10 @@
1
1
  import {
2
2
  runCampaign
3
- } from "./chunk-ZZ2HOPME.js";
4
- import "./chunk-IDVBLYCY.js";
3
+ } from "./chunk-TWS7AZEY.js";
4
+ import "./chunk-LMZQ2Z4U.js";
5
5
  import "./chunk-3BFEG2F6.js";
6
6
  import "./chunk-PZ5AY32C.js";
7
7
  export {
8
8
  runCampaign
9
9
  };
10
- //# sourceMappingURL=run-campaign-4Y5V5CN3.js.map
10
+ //# sourceMappingURL=run-campaign-RDGAM5KJ.js.map
@@ -0,0 +1 @@
1
+ {"version":3,"sources":[],"sourcesContent":[],"mappings":"","names":[]}
@@ -1,5 +1,5 @@
1
- import { L as LlmClientOptions } from './llm-client-CuUg2Mn3.js';
2
- import { I as ImprovementDriver, S as Scenario, g as CampaignResult, k as GateResult, D as DispatchFn, a as JudgeConfig, L as LabeledScenarioStore, h as CampaignTraceWriter, m as GenerationRecord, M as MutableSurface, P as ParetoParent, G as Gate } from './types-D7lLRYe9.js';
1
+ import { L as LlmClientOptions } from './llm-client-BeEcAokY.js';
2
+ import { I as ImprovementDriver, S as Scenario, g as CampaignResult, k as GateResult, D as DispatchFn, a as JudgeConfig, L as LabeledScenarioStore, h as CampaignTraceWriter, m as GenerationRecord, M as MutableSurface, P as ParetoParent, G as Gate } from './types-BU-7W85F.js';
3
3
 
4
4
  /**
5
5
  * @experimental
@@ -1,4 +1,4 @@
1
- import { V as ValidationError } from './errors-Dwqw-T_m.js';
1
+ import { V as ValidationError } from './errors-CzMUYo7b.js';
2
2
  import { F as FailureClass } from './schema-m0gsnbt3.js';
3
3
 
4
4
  type AgentProfileCellSchemaVersion = 'agent-profile-cell/v1';
@@ -1,4 +1,4 @@
1
- import { a as RunSplitTag } from './run-record-De9VarXR.js';
1
+ import { a as RunSplitTag } from './run-record-e7vj1uZQ.js';
2
2
 
3
3
  interface RuntimeTrajectoryHookEvent {
4
4
  id: string;
@@ -1,11 +1,10 @@
1
1
  import { AxAIService } from '@ax-llm/ax';
2
- import { c as TraceAnalystKindSpec } from './kind-factory-CVecZZG_.js';
3
- import { b as AnalystRegistryOptions, a as AnalystRegistry } from './registry-DrEQ3Luj.js';
4
2
  import { z } from 'zod';
5
- import { c as AnalystFinding, A as Analyst, a as AnalystContext } from './types-Cu3u_x59.js';
3
+ import { c as AnalystFinding, A as Analyst, a as AnalystContext } from './types-2VVIL04s.js';
4
+ import { T as TraceAnalystKindSpec } from './kind-factory-5b7xXXOr.js';
6
5
  import { a as TraceAnalystSpan } from './store-C1YxJDEK.js';
7
- import { L as LlmClientOptions } from './llm-client-CuUg2Mn3.js';
8
- import { S as Severity } from './multi-layer-verifier-DlWCXuxL.js';
6
+ import { L as LlmClientOptions } from './llm-client-BeEcAokY.js';
7
+ import { S as Severity } from './multi-layer-verifier-DUZXrPDA.js';
9
8
 
10
9
  interface CreateAnalystAiConfig {
11
10
  /** OpenAI-compatible API key forwarded as `Authorization: Bearer`.
@@ -33,31 +32,6 @@ interface CreateAnalystAiConfig {
33
32
  */
34
33
  declare function createAnalystAi(config: CreateAnalystAiConfig): AxAIService;
35
34
 
36
- /**
37
- * `buildDefaultAnalystRegistry` — the canonical analyst suite, so consumers
38
- * stop hand-wiring `new AnalystRegistry()` + per-kind `createTraceAnalystKind`.
39
- *
40
- * The deterministic `behavioralAnalyst` is ALWAYS registered (it needs no
41
- * model and is model-agnostic by construction). The agentic RLM kinds are
42
- * registered only when an `ai` service is supplied — so a caller with no LLM
43
- * still gets the full behavioral/efficiency diagnosis, and the substrate's
44
- * "any model (including no model)" guarantee holds at the suite level.
45
- */
46
-
47
- interface DefaultAnalystRegistryOptions {
48
- /** Ax service for the agentic RLM kinds. Omit → only the deterministic analyst. */
49
- ai?: AxAIService;
50
- /** Model for the agentic kinds (falls back to the ai service default). */
51
- model?: string;
52
- /** Which agentic kinds to register when `ai` is present. Default = the shipped suite. */
53
- kinds?: readonly TraceAnalystKindSpec[];
54
- /** Set false to omit the deterministic behavioral analyst (default: include). */
55
- includeBehavioral?: boolean;
56
- /** Forwarded to the AnalystRegistry constructor (signal, tags, priorFindings). */
57
- registry?: AnalystRegistryOptions;
58
- }
59
- declare function buildDefaultAnalystRegistry(opts?: DefaultAnalystRegistryOptions): AnalystRegistry;
60
-
61
35
  /**
62
36
  * Typed `FindingSubject` — the canonical grammar every analyst kind emits.
63
37
  *
@@ -647,4 +621,4 @@ declare function runSemanticConceptJudge(input: SemanticConceptJudgeInput, optio
647
621
  */
648
622
  declare function createSemanticConceptJudge(options?: SemanticConceptJudgeOptions): (input: SemanticConceptJudgeInput) => Promise<SemanticConceptJudgeResult>;
649
623
 
650
- export { type ConceptFinding as A, type BehavioralMetrics as B, type CreateAnalystAiConfig as C, DEFAULT_TRACE_ANALYST_KINDS as D, type ConceptSpec as E, FAILURE_MODE_KIND_SPEC as F, type ConceptWeightStrategy as G, DEFAULT_COMPLEXITY_WEIGHTS as H, IMPROVEMENT_KIND_SPEC as I, SEMANTIC_CONCEPT_JUDGE_VERSION as J, KIND_EXPECTED_SUBJECTS as K, type SemanticConceptJudgeResult as L, type SuboptimalCode as M, type SuboptimalSignal as N, computeTraceMetrics as O, type PersistedFinding as P, createSemanticConceptJudge as Q, runSemanticConceptJudge as R, type SemanticConceptJudgeOptions as S, type SemanticConceptJudgeInput as a, type DefaultAnalystRegistryOptions as b, type DiffPolicy as c, FINDING_SUBJECT_GRAMMAR_PROMPT as d, FINDING_SUBJECT_KINDS as e, type FindingSubject as f, type FindingSubjectKind as g, FindingSubjectStringSchema as h, type FindingsDiff as i, FindingsStore as j, KNOWLEDGE_GAP_KIND_SPEC as k, KNOWLEDGE_POISONING_KIND_SPEC as l, SKILL_USAGE_ANALYST as m, SkillUsageAnalyst as n, type SkillUsageRecord as o, type SkillUsageReport as p, type SkillUsageScanConfig as q, buildDefaultAnalystRegistry as r, buildSkillUsageReport as s, createAnalystAi as t, defaultIsMaterial as u, diffFindings as v, emitSkillUsageFindings as w, parseFindingSubject as x, renderFindingSubject as y, type ConceptComplexity as z };
624
+ export { type ConceptWeightStrategy as A, type BehavioralMetrics as B, type CreateAnalystAiConfig as C, DEFAULT_TRACE_ANALYST_KINDS as D, DEFAULT_COMPLEXITY_WEIGHTS as E, FAILURE_MODE_KIND_SPEC as F, SEMANTIC_CONCEPT_JUDGE_VERSION as G, type SemanticConceptJudgeResult as H, IMPROVEMENT_KIND_SPEC as I, type SuboptimalCode as J, KIND_EXPECTED_SUBJECTS as K, type SuboptimalSignal as L, computeTraceMetrics as M, createSemanticConceptJudge as N, runSemanticConceptJudge as O, type PersistedFinding as P, type SemanticConceptJudgeOptions as S, type SemanticConceptJudgeInput as a, type DiffPolicy as b, FINDING_SUBJECT_GRAMMAR_PROMPT as c, FINDING_SUBJECT_KINDS as d, type FindingSubject as e, type FindingSubjectKind as f, FindingSubjectStringSchema as g, type FindingsDiff as h, FindingsStore as i, KNOWLEDGE_GAP_KIND_SPEC as j, KNOWLEDGE_POISONING_KIND_SPEC as k, SKILL_USAGE_ANALYST as l, SkillUsageAnalyst as m, type SkillUsageRecord as n, type SkillUsageReport as o, type SkillUsageScanConfig as p, buildSkillUsageReport as q, createAnalystAi as r, defaultIsMaterial as s, diffFindings as t, emitSkillUsageFindings as u, parseFindingSubject as v, renderFindingSubject as w, type ConceptComplexity as x, type ConceptFinding as y, type ConceptSpec as z };
@@ -0,0 +1,33 @@
1
+ /**
2
+ * Series convergence — detects whether a sequence of scalar measurements
3
+ * is stabilizing, drifting, or noisy.
4
+ *
5
+ * Lifted from ADC convergence.ts. The per-turn `ConvergenceTracker` is
6
+ * about progress *within* a single run; this module is about drift
7
+ * *across* runs (e.g. "are my nightly eval scores stabilizing?").
8
+ *
9
+ * Three signals:
10
+ * - stabilized: last K values have low variance (< epsilon) — done
11
+ * - drifting: recent trend is monotonic and beyond noise — regressing or improving
12
+ * - noisy: neither — keep iterating, but flag as untrustworthy for gating
13
+ */
14
+ interface SeriesConvergenceOptions {
15
+ /** Window size for "recent" analysis (default 5). */
16
+ window?: number;
17
+ /** Coefficient-of-variation threshold below which the window is stabilized (default 0.05 = 5%). */
18
+ stableCv?: number;
19
+ /** Minimum monotone run length to call drift (default 3). */
20
+ driftRun?: number;
21
+ }
22
+ interface SeriesConvergenceResult {
23
+ state: 'stabilized' | 'drifting-up' | 'drifting-down' | 'noisy' | 'insufficient-data';
24
+ windowMean: number;
25
+ windowCv: number;
26
+ /** Longest monotonic run at the tail of the series (positive for up, negative for down). */
27
+ tailRun: number;
28
+ /** True when n ≥ window AND windowCv ≤ stableCv. */
29
+ stable: boolean;
30
+ }
31
+ declare function analyzeSeries(values: number[], options?: SeriesConvergenceOptions): SeriesConvergenceResult;
32
+
33
+ export { type SeriesConvergenceOptions as S, type SeriesConvergenceResult as a, analyzeSeries as b };
@@ -1,4 +1,4 @@
1
- import { C as ContinuousAgreementOptions, a as ContinuousAgreement } from './judge-calibration-DilmB3Ml.js';
1
+ import { d as ContinuousAgreementOptions, C as ContinuousAgreement } from './judge-calibration-0p2QcWNE.js';
2
2
  import { J as JudgeScore } from './types-Croy5h7V.js';
3
3
 
4
4
  /** Identity: dimensions already follow "higher = better" by prompt convention
@@ -245,5 +245,74 @@ interface PairedBootstrapOptions {
245
245
  * gain is real at the confidence level. Throws on unequal sample sizes.
246
246
  */
247
247
  declare function pairedBootstrap(before: number[], after: number[], opts?: PairedBootstrapOptions): PairedBootstrapResult;
248
+ interface EProcessOptions {
249
+ /** Type-I error budget. The process decides when wealth ≥ 1/alpha
250
+ * (Ville's inequality). Default 0.05. */
251
+ alpha?: number;
252
+ /** Truncation bound on the predictable bet λ ∈ [0, maxBet]. Must satisfy
253
+ * maxBet < 1/nullMean so every wealth factor stays strictly positive.
254
+ * Default 0.5. */
255
+ maxBet?: number;
256
+ /** The null boundary m₀ for H0: E[x] ≤ m₀ on x ∈ [0,1]. Default 0.5
257
+ * (the paired-delta encoding x = (d+1)/2 maps "no effect" to 1/2).
258
+ * A pre-registered minEffect shifts this — see `sequentialPairedGate`. */
259
+ nullMean?: number;
260
+ }
261
+ interface EProcessStep {
262
+ /** Current wealth W_n — the e-value against H0 after n observations. */
263
+ wealth: number;
264
+ /** Observations consumed so far. */
265
+ n: number;
266
+ /** True from the first n where W_n ≥ 1/alpha onward (sticky). */
267
+ decided: boolean;
268
+ }
269
+ interface EProcessState extends EProcessStep {
270
+ alpha: number;
271
+ maxBet: number;
272
+ nullMean: number;
273
+ /** The decision boundary 1/alpha. */
274
+ threshold: number;
275
+ /** Observation count at the first threshold crossing; undefined until decided. */
276
+ decidedAtN?: number;
277
+ }
278
+ interface EProcess {
279
+ /** Consume one observation x ∈ [0,1]. Throws on non-finite / out-of-range
280
+ * input — a silent clamp would corrupt the type-I guarantee. */
281
+ update(x: number): EProcessStep;
282
+ state(): EProcessState;
283
+ }
284
+ /**
285
+ * Betting test-martingale for bounded observations — the e-process core of
286
+ * anytime-valid sequential testing (Waudby-Smith & Ramdas, "Estimating means
287
+ * of bounded random variables by betting", JRSS-B 2024).
288
+ *
289
+ * Observations x_i ∈ [0,1]; H0: E[x] ≤ m₀ (`nullMean`, default 1/2). Wealth
290
+ *
291
+ * W_t = Π_{i≤t} (1 + λ_i (x_i − m₀)), W_0 = 1
292
+ *
293
+ * with the truncated GROW-style plug-in bet computed from PRIOR observations:
294
+ *
295
+ * λ_i = clamp((μ̂_{i−1} − m₀) / (σ̂²_{i−1} + (μ̂_{i−1} − m₀)²), 0, maxBet)
296
+ *
297
+ * where μ̂/σ̂² are the shrunk running estimates μ̂_t = (1/2 + Σx_i)/(t+1),
298
+ * σ̂²_t = (1/4 + Σ(x_i − μ̂_i)²)/(t+1).
299
+ *
300
+ * PREDICTABILITY INVARIANT (load-bearing): λ_i is a function of x_1..x_{i−1}
301
+ * ONLY — it may never see x_i. With λ_i ≥ 0 predictable, each factor has
302
+ * E[1 + λ_i(x_i − m₀) | past] ≤ 1 under H0, so W is a nonnegative
303
+ * supermartingale and Ville's inequality gives P(∃t: W_t ≥ 1/α) ≤ α — the
304
+ * type-I guarantee holds at ANY data-dependent stopping time. λ_1 is always 0
305
+ * (no prior evidence), so the first observation never moves wealth.
306
+ *
307
+ * `decided` latches at the first crossing W_t ≥ 1/α and never un-latches;
308
+ * wealth keeps updating after the crossing (the e-process remains valid), but
309
+ * the decision time is the first crossing.
310
+ */
311
+ declare function eProcess(opts?: EProcessOptions): EProcess;
312
+ /** Tiny seedable PRNG (mulberry32) — deterministic resampling/shuffling, not
313
+ * cryptographic. Exported so e-process shuffles and bootstrap resampling
314
+ * share ONE PRNG implementation; a seed is REQUIRED (unseeded randomness in
315
+ * gate verdicts is non-reproducible by construction). */
316
+ declare function mulberry32(seed: number): () => number;
248
317
 
249
- export { type CliffsMagnitude as C, type PairedBootstrapOptions as P, type WeightedCompositeInput as W, type PairedBootstrapResult as a, benjaminiHochberg as b, type CorpusAgreementOptions as c, type CorpusAgreementPerDimension as d, type CorpusAgreementReport as e, type CorpusScoreRecord as f, type WeightedCompositeResult as g, bonferroni as h, cliffsDelta as i, cohensD as j, confidenceInterval as k, corpusInterRaterAgreement as l, corpusInterRaterAgreementFromJudgeScores as m, interRaterReliability as n, interpretCliffs as o, pairedBootstrap as p, mannWhitneyU as q, normalizeScores as r, pairedMde as s, pairedTTest as t, partialCredit as u, requiredSampleSize as v, wilcoxonSignedRank as w, weightedComposite as x, weightedMean as y };
318
+ export { partialCredit as A, requiredSampleSize as B, type CorpusAgreementReport as C, weightedComposite as D, type EProcessState as E, weightedMean as F, type PairedBootstrapOptions as P, type WeightedCompositeInput as W, type PairedBootstrapResult as a, benjaminiHochberg as b, type CliffsMagnitude as c, type CorpusAgreementOptions as d, type CorpusAgreementPerDimension as e, type CorpusScoreRecord as f, type EProcess as g, type EProcessOptions as h, type EProcessStep as i, type WeightedCompositeResult as j, bonferroni as k, cliffsDelta as l, cohensD as m, confidenceInterval as n, corpusInterRaterAgreement as o, pairedBootstrap as p, corpusInterRaterAgreementFromJudgeScores as q, eProcess as r, interRaterReliability as s, interpretCliffs as t, mannWhitneyU as u, mulberry32 as v, wilcoxonSignedRank as w, normalizeScores as x, pairedMde as y, pairedTTest as z };
@@ -1,4 +1,4 @@
1
- import { R as RunRecord } from './run-record-De9VarXR.js';
1
+ import { R as RunRecord } from './run-record-e7vj1uZQ.js';
2
2
  import { F as FailureClusterReport } from './failure-cluster-CL7IVgkJ.js';
3
3
 
4
4
  /**
package/dist/traces.d.ts CHANGED
@@ -1,9 +1,9 @@
1
- import { N as NotFoundError, R as ReplayError } from './errors-Dwqw-T_m.js';
1
+ import { N as NotFoundError, R as ReplayError } from './errors-CzMUYo7b.js';
2
2
  import { P as ProviderRedactor, R as RawProviderSink, d as RawProviderEvent } from './raw-provider-sink-C46HDghv.js';
3
3
  export { F as FileSystemRawProviderSink, a as FileSystemRawProviderSinkOptions, I as InMemoryRawProviderSink, b as InMemoryRawProviderSinkOptions, N as NoopRawProviderSink, c as RawProviderDirection, e as RawProviderSinkFilter, f as defaultProviderRedactor, p as providerFromBaseUrl } from './raw-provider-sink-C46HDghv.js';
4
4
  import { a as RunCompleteHookContext, R as RunCompleteHook } from './emitter-DEZwY14K.js';
5
5
  export { S as SpanHandle, T as TraceEmitter, b as TraceEmitterOptions, l as llmSpanFromProvider } from './emitter-DEZwY14K.js';
6
- export { b as RunIntegrityError, R as RunIntegrityExpectations, c as RunIntegrityIssue, d as RunIntegrityIssueCode, a as RunIntegrityReport, e as assertRunCaptured, t as throwIfRunIncomplete } from './integrity-CJzrpUua.js';
6
+ export { b as RunIntegrityError, R as RunIntegrityExpectations, c as RunIntegrityIssue, d as RunIntegrityIssueCode, a as RunIntegrityReport, e as assertRunCaptured, t as throwIfRunIncomplete } from './integrity-VJ9A7aST.js';
7
7
  import { T as TraceStore } from './store-CKUAgsJz.js';
8
8
  export { E as EventFilter, F as FileSystemTraceStore, a as FileSystemTraceStoreOptions, I as InMemoryTraceStore, R as RunFilter, S as SpanFilter } from './store-CKUAgsJz.js';
9
9
  export { a as aggregateLlm, b as argHash, g as groupBy, j as judgeSpans, l as llmSpans, r as runFailureClass, c as runsForScenario, t as toolSpans } from './query-CqTxMwDw.js';
@@ -14,7 +14,7 @@ import { A as AnalyzeTracesOptions, b as AnalyzeTracesResult } from './analyst-C
14
14
  export { a as AnalyzeTracesInput, c as AnalyzeTracesTurnSnapshot, d as analyzeTraces } from './analyst-C8HHvfJp.js';
15
15
  import { h as TraceAnalystSpanKind, i as TraceAnalystSpanStatus, T as TraceAnalysisStore, g as TraceAnalystFilters, b as DatasetOverview, Q as QueryTracesPage, l as ViewTraceResult, V as ViewSpansResult, c as SearchTraceResult, S as SearchSpanResult } from './store-C1YxJDEK.js';
16
16
  export { D as DEFAULT_TRACE_ANALYST_BUDGETS, E as ErrorCluster, d as SpanMatchRecord, e as TRACE_ANALYST_TRUNCATION_MARKER_PREFIX, f as TraceAnalystByteBudgets, a as TraceAnalystSpan, j as TraceAnalystTraceSummary, k as ViewTraceOversized } from './store-C1YxJDEK.js';
17
- import { a as RunSplitTag, b as RunTokenUsage, R as RunRecord } from './run-record-De9VarXR.js';
17
+ import { a as RunSplitTag, b as RunTokenUsage, R as RunRecord } from './run-record-e7vj1uZQ.js';
18
18
  import { AxFunction } from '@ax-llm/ax';
19
19
 
20
20
  /**
package/dist/traces.js CHANGED
@@ -28,7 +28,7 @@ import {
28
28
  scoreTraceInsightReadiness,
29
29
  tokenizeDomainWords,
30
30
  traceAnalystOnRunComplete
31
- } from "./chunk-P2J6SOXT.js";
31
+ } from "./chunk-QG2OVF2D.js";
32
32
  import {
33
33
  DEFAULT_REDACTION_RULES,
34
34
  REDACTION_VERSION,
@@ -55,16 +55,18 @@ import {
55
55
  isToolSpan
56
56
  } from "./chunk-5BKGXME7.js";
57
57
  import {
58
- DEFAULT_TRACE_ANALYST_BUDGETS,
59
- OtlpFileTraceStore,
60
- SpanNotFoundError,
61
58
  TRACE_ANALYST_ACTOR_DESCRIPTION,
62
59
  TRACE_ANALYST_ACTOR_DESCRIPTION_VERSION,
63
60
  TRACE_ANALYST_SUBAGENT_DESCRIPTION,
61
+ analyzeTraces
62
+ } from "./chunk-UHMJT4T7.js";
63
+ import {
64
+ DEFAULT_TRACE_ANALYST_BUDGETS,
65
+ OtlpFileTraceStore,
66
+ SpanNotFoundError,
64
67
  TRACE_ANALYST_TRUNCATION_MARKER_PREFIX,
65
68
  TraceFileMissingError,
66
69
  TraceNotFoundError,
67
- analyzeTraces,
68
70
  asNumber,
69
71
  asString,
70
72
  buildTraceAnalystTools,
@@ -76,7 +78,7 @@ import {
76
78
  readOtlpStatus,
77
79
  stringField,
78
80
  traceAnalystFunctionGroup
79
- } from "./chunk-VUINJM5M.js";
81
+ } from "./chunk-QAY5UIJO.js";
80
82
  import {
81
83
  RunIntegrityError,
82
84
  assertRunCaptured,
@@ -1,7 +1,7 @@
1
- import { R as RunRecord } from './run-record-De9VarXR.js';
1
+ import { R as RunRecord } from './run-record-e7vj1uZQ.js';
2
2
  import { T as TraceAnalysisStore } from './store-C1YxJDEK.js';
3
3
  import { a as JudgeInput } from './types-Croy5h7V.js';
4
- import { b as LlmCallRequest, c as LlmCallResult } from './llm-client-CuUg2Mn3.js';
4
+ import { b as LlmCallRequest, c as LlmCallResult } from './llm-client-BeEcAokY.js';
5
5
 
6
6
  /**
7
7
  * ChatClient — the single LLM abstraction analysts call.
@@ -1,4 +1,4 @@
1
- import { b as RunTokenUsage } from './run-record-De9VarXR.js';
1
+ import { b as RunTokenUsage } from './run-record-e7vj1uZQ.js';
2
2
 
3
3
  /**
4
4
  * @experimental
@@ -93,10 +93,30 @@ interface JudgeConfig<TArtifact, TScenario extends Scenario = Scenario> {
93
93
  }): JudgeScore | Promise<JudgeScore>;
94
94
  appliesTo?: (scenario: TScenario) => boolean;
95
95
  }
96
+ /** The canonical judge verdict shape — one declaration, shared by campaign
97
+ * judges and the multishot judge runner (which re-exports this type).
98
+ *
99
+ * Scale is PRODUCER-DEFINED: campaign convention is [0,1]; the legacy
100
+ * multishot runner emits 0-10. Cross-scale comparison must go through
101
+ * `detectScale` (src/campaign/gates/statistical-heldout.ts, used by
102
+ * promotion-policy) — never renormalize a producer's values in place, as
103
+ * downstream thresholds (`composite >= 5` in multishot/matrix.ts, live-soak
104
+ * `>= 7` gates) key on the producer's native scale. */
96
105
  interface JudgeScore {
97
106
  dimensions: Record<string, number>;
98
107
  composite: number;
99
108
  notes: string;
109
+ /** Set when the judge itself failed (call error, unparseable output).
110
+ * `composite`/`dimensions` carry no signal — aggregators MUST exclude
111
+ * failed scores from means instead of folding them into zeros. */
112
+ failed?: true;
113
+ /** Ensemble extras (populated by `ensembleJudge`): max per-dimension
114
+ * spread across surviving judges — the inter-rater signal. */
115
+ maxDisagreement?: number;
116
+ /** Ensemble extras: judge identities whose verdict failed. */
117
+ failedJudges?: string[];
118
+ /** Ensemble extras: each surviving judge's per-dimension scores. */
119
+ perJudge?: Record<string, Record<string, number>>;
100
120
  }
101
121
  /** @experimental A tier-4 code surface — a candidate change to the agent's
102
122
  * IMPLEMENTATION, not its prompt. Produced by autoresearch (reads codebase +
@@ -1,4 +1,4 @@
1
- import { D as DefaultVerdict } from './verdict-CeEgtjyI.js';
1
+ import { D as DefaultVerdict } from './verdict-C9MlYujm.js';
2
2
 
3
3
  /**
4
4
  * @experimental
@@ -17,6 +17,9 @@
17
17
  * Minimal verdict shape — `valid` + `score` are required; `scores` +
18
18
  * `notes` are optional surface. Validators that need richer shapes
19
19
  * parameterise `Validator<Output, MyVerdict>` with their own type.
20
+ *
21
+ * Need structured extras? Extend DefaultVerdict with typed fields — never
22
+ * serialize extras into `notes`.
20
23
  */
21
24
  interface DefaultVerdict {
22
25
  /** Whether the output meets the validator's pass criteria. */
@@ -1,4 +1,4 @@
1
- import { F as FeedbackTrajectoryStore } from '../feedback-trajectory-B3rErRsh.js';
1
+ import { F as FeedbackTrajectoryStore } from '../feedback-trajectory-D9OVLrg9.js';
2
2
  import { T as TraceStore } from '../store-CKUAgsJz.js';
3
3
  import { z } from 'zod';
4
4
  import { OpenAPIObject } from 'openapi3-ts/oas31';
@@ -8,8 +8,8 @@ import { Hono } from 'hono';
8
8
  import '../control-runtime-DuFBYg7A.js';
9
9
  import '../emitter-DEZwY14K.js';
10
10
  import '../schema-m0gsnbt3.js';
11
- import '../dataset-B2kL-fSM.js';
12
- import '../errors-Dwqw-T_m.js';
11
+ import '../dataset-BbGkaN2I.js';
12
+ import '../errors-CzMUYo7b.js';
13
13
 
14
14
  declare const RubricDimensionSchema: z.ZodObject<{
15
15
  id: z.ZodString;
@@ -1,25 +1,26 @@
1
1
  import { W as WorkflowTopology } from '../harness-optimizer-EnEnQPsr.js';
2
- import { a as RunSplitTag, b as RunTokenUsage, R as RunRecord } from '../run-record-De9VarXR.js';
3
- import { c as AnalystFinding, h as AnalystSeverity, E as EvidenceRef } from '../types-Cu3u_x59.js';
4
- import { F as FailureClusterInsight } from '../insight-report-3ADTfClO.js';
5
- import { a as VerificationReport, L as LayerResult } from '../multi-layer-verifier-DlWCXuxL.js';
2
+ import { a as RunSplitTag, b as RunTokenUsage, R as RunRecord } from '../run-record-e7vj1uZQ.js';
3
+ import { c as AnalystFinding, h as AnalystSeverity, E as EvidenceRef } from '../types-2VVIL04s.js';
4
+ import { F as FailureClusterInsight } from '../insight-report-BBwvOh6x.js';
5
+ import { a as VerificationReport, L as LayerResult } from '../multi-layer-verifier-DUZXrPDA.js';
6
6
  import { F as FailureClusterReport } from '../failure-cluster-CL7IVgkJ.js';
7
7
  import { R as RedactionRule, a as RedactionReport } from '../redact-B40YG2M_.js';
8
- import { D as DatasetSplit } from '../dataset-B2kL-fSM.js';
9
- import { a as FeedbackTrajectory } from '../feedback-trajectory-B3rErRsh.js';
10
- import { a as PairedBootstrapResult } from '../statistics-CnC1FMbx.js';
8
+ import { D as DatasetSplit } from '../dataset-BbGkaN2I.js';
9
+ import { a as FeedbackTrajectory } from '../feedback-trajectory-D9OVLrg9.js';
10
+ import { a as PairedBootstrapResult } from '../statistics-C7PozGrZ.js';
11
11
  import '../pareto-E-pembql.js';
12
12
  import '../run-critic-BAIjX99r.js';
13
13
  import '../schema-m0gsnbt3.js';
14
14
  import '../store-CKUAgsJz.js';
15
- import '../errors-Dwqw-T_m.js';
15
+ import '../errors-CzMUYo7b.js';
16
16
  import '../store-C1YxJDEK.js';
17
17
  import '../types-Croy5h7V.js';
18
18
  import '@tangle-network/tcloud';
19
- import '../llm-client-CuUg2Mn3.js';
19
+ import '../llm-client-BeEcAokY.js';
20
20
  import '../raw-provider-sink-C46HDghv.js';
21
- import '../summary-report-Db0dDSWP.js';
22
- import '../judge-calibration-DilmB3Ml.js';
21
+ import '../summary-report-DGmUucwQ.js';
22
+ import '../judge-calibration-0p2QcWNE.js';
23
+ import '../verdict-C9MlYujm.js';
23
24
  import '../control-runtime-DuFBYg7A.js';
24
25
  import '../emitter-DEZwY14K.js';
25
26
 
@@ -1,6 +1,6 @@
1
1
  import {
2
2
  pairedBootstrap
3
- } from "../chunk-IDVBLYCY.js";
3
+ } from "../chunk-LMZQ2Z4U.js";
4
4
  import {
5
5
  DEFAULT_REDACTION_RULES,
6
6
  redactString
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "@tangle-network/agent-eval",
3
- "version": "0.86.0",
3
+ "version": "0.90.0",
4
4
  "description": "Evaluate and improve AI agents from runs, traces, judges, and feedback. Compare candidates, cluster failures, measure lift, and gate releases.",
5
5
  "homepage": "https://github.com/tangle-network/agent-eval#readme",
6
6
  "repository": {
@@ -39,6 +39,16 @@
39
39
  "import": "./dist/rl.js",
40
40
  "default": "./dist/rl.js"
41
41
  },
42
+ "./diagnose": {
43
+ "types": "./dist/diagnose.d.ts",
44
+ "import": "./dist/diagnose.js",
45
+ "default": "./dist/diagnose.js"
46
+ },
47
+ "./fuzz": {
48
+ "types": "./dist/fuzz.d.ts",
49
+ "import": "./dist/fuzz.js",
50
+ "default": "./dist/fuzz.js"
51
+ },
42
52
  "./traces": {
43
53
  "types": "./dist/traces.d.ts",
44
54
  "import": "./dist/traces.js",
@@ -99,6 +109,11 @@
99
109
  "import": "./dist/matrix/index.js",
100
110
  "default": "./dist/matrix/index.js"
101
111
  },
112
+ "./perf": {
113
+ "types": "./dist/perf/index.d.ts",
114
+ "import": "./dist/perf/index.js",
115
+ "default": "./dist/perf/index.js"
116
+ },
102
117
  "./multishot": {
103
118
  "types": "./dist/multishot/index.d.ts",
104
119
  "import": "./dist/multishot/index.js",