@tangle-network/agent-eval 0.117.1 → 0.118.1

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (143) hide show
  1. package/CHANGELOG.md +30 -0
  2. package/dist/analyst/index.d.ts +2772 -21
  3. package/dist/analyst/index.js +7 -6
  4. package/dist/analyst/index.js.map +1 -1
  5. package/dist/belief-state/index.d.ts +706 -10
  6. package/dist/belief-state/index.js +2 -1
  7. package/dist/belief-state/index.js.map +1 -1
  8. package/dist/benchmarks/index.d.ts +958 -14
  9. package/dist/benchmarks/index.js +12 -10
  10. package/dist/builder-eval/index.d.ts +449 -4
  11. package/dist/builder-eval/index.js +4 -3
  12. package/dist/builder-eval/index.js.map +1 -1
  13. package/dist/campaign/index.d.ts +4275 -73
  14. package/dist/campaign/index.js +12 -10
  15. package/dist/{chunk-VF3XSYTI.js → chunk-33JA4TFA.js} +6 -6
  16. package/dist/{chunk-4JLWXDYA.js → chunk-3EHHMC6E.js} +2 -2
  17. package/dist/{chunk-CCZIVI3F.js → chunk-BFW56GTT.js} +2 -2
  18. package/dist/{chunk-YZPO4UHR.js → chunk-FTUMG2U7.js} +124 -149
  19. package/dist/chunk-FTUMG2U7.js.map +1 -0
  20. package/dist/{chunk-E4BUPP7Z.js → chunk-HKUCJ437.js} +38 -66
  21. package/dist/chunk-HKUCJ437.js.map +1 -0
  22. package/dist/{chunk-HZHNRYHK.js → chunk-K6N6XJJX.js} +2 -2
  23. package/dist/chunk-KSDQVPLR.js +286 -0
  24. package/dist/chunk-KSDQVPLR.js.map +1 -0
  25. package/dist/chunk-MA6HLL3S.js +65 -0
  26. package/dist/chunk-MA6HLL3S.js.map +1 -0
  27. package/dist/{chunk-MGEHEHSN.js → chunk-OIIMMLRB.js} +11 -11
  28. package/dist/{chunk-DXZRATT5.js → chunk-OYZAPX5G.js} +3 -3
  29. package/dist/chunk-PXE2VKMX.js +140 -0
  30. package/dist/chunk-PXE2VKMX.js.map +1 -0
  31. package/dist/{chunk-JSJZ4PJ6.js → chunk-Q442S5AS.js} +17 -17
  32. package/dist/{chunk-ODVOOEWQ.js → chunk-QBRSJK47.js} +2 -2
  33. package/dist/{chunk-S2F4J57L.js → chunk-QKEGNI5B.js} +77 -32
  34. package/dist/chunk-QKEGNI5B.js.map +1 -0
  35. package/dist/{chunk-5UF54T55.js → chunk-S3UZOQ5Y.js} +34 -7
  36. package/dist/chunk-S3UZOQ5Y.js.map +1 -0
  37. package/dist/{chunk-FQNLDL4D.js → chunk-SVH2ANFD.js} +136 -4
  38. package/dist/chunk-SVH2ANFD.js.map +1 -0
  39. package/dist/{chunk-GQCZRZ7L.js → chunk-U5CHZ5M3.js} +9 -9
  40. package/dist/{chunk-TVVP3ZZQ.js → chunk-VQMK5FMP.js} +2 -1
  41. package/dist/{chunk-TVVP3ZZQ.js.map → chunk-VQMK5FMP.js.map} +1 -1
  42. package/dist/chunk-WDHBCA3M.js +31 -0
  43. package/dist/chunk-WDHBCA3M.js.map +1 -0
  44. package/dist/{chunk-HQPHZGL6.js → chunk-YLMUS4MM.js} +9 -9
  45. package/dist/{chunk-LQUTGLOZ.js → chunk-ZET2UAYW.js} +15 -65
  46. package/dist/chunk-ZET2UAYW.js.map +1 -0
  47. package/dist/contract/index.d.ts +4012 -38
  48. package/dist/contract/index.js +56 -29
  49. package/dist/contract/index.js.map +1 -1
  50. package/dist/control.d.ts +1013 -9
  51. package/dist/control.js +4 -3
  52. package/dist/fuzz.d.ts +194 -4
  53. package/dist/fuzz.js +3 -3
  54. package/dist/hosted/index.d.ts +498 -17
  55. package/dist/index.d.ts +10992 -1288
  56. package/dist/index.js +101 -80
  57. package/dist/index.js.map +1 -1
  58. package/dist/matrix/index.d.ts +139 -4
  59. package/dist/meta-eval/index.d.ts +862 -15
  60. package/dist/meta-eval/index.js +2 -1
  61. package/dist/meta-eval/index.js.map +1 -1
  62. package/dist/multishot/index.d.ts +214 -14
  63. package/dist/openapi.json +1 -1
  64. package/dist/pipelines/index.d.ts +392 -7
  65. package/dist/pipelines/index.js +5 -3
  66. package/dist/pipelines/index.js.map +1 -1
  67. package/dist/reporting.d.ts +1277 -17
  68. package/dist/rl.d.ts +2359 -28
  69. package/dist/rl.js +6 -5
  70. package/dist/rl.js.map +1 -1
  71. package/dist/storyboard/index.d.ts +86 -1
  72. package/dist/trace-attributes.d.ts +16 -0
  73. package/dist/trace-attributes.js +32 -0
  74. package/dist/trace-attributes.js.map +1 -0
  75. package/dist/traces.d.ts +1978 -697
  76. package/dist/traces.js +53 -32
  77. package/dist/wire/index.d.ts +655 -9
  78. package/docs/insight-report.md +44 -0
  79. package/package.json +9 -3
  80. package/dist/adversarial-B7loGVVX.d.ts +0 -19
  81. package/dist/analyst-C8HHvfJp.d.ts +0 -88
  82. package/dist/analyze-runs--2x39HZ7.d.ts +0 -81
  83. package/dist/baseline-DKq3gJpP.d.ts +0 -141
  84. package/dist/calibration-C8MTS7cw.d.ts +0 -101
  85. package/dist/chunk-5UF54T55.js.map +0 -1
  86. package/dist/chunk-E4BUPP7Z.js.map +0 -1
  87. package/dist/chunk-FQNLDL4D.js.map +0 -1
  88. package/dist/chunk-LQUTGLOZ.js.map +0 -1
  89. package/dist/chunk-S2F4J57L.js.map +0 -1
  90. package/dist/chunk-YZPO4UHR.js.map +0 -1
  91. package/dist/code-agent-session-CjZsVd19.d.ts +0 -87
  92. package/dist/control-6vuGfmDH.d.ts +0 -258
  93. package/dist/cost-ledger-DWy3XdJc.d.ts +0 -183
  94. package/dist/dataset-NENEzRgk.d.ts +0 -115
  95. package/dist/default-registry-DaK8b3fv.d.ts +0 -155
  96. package/dist/emitter-CjD7vUwv.d.ts +0 -122
  97. package/dist/errors-oeQrLqXC.d.ts +0 -74
  98. package/dist/failure-cluster-DOAcSJ87.d.ts +0 -76
  99. package/dist/feedback-trajectory-BUnM58xL.d.ts +0 -348
  100. package/dist/gepa-eESocoDi.d.ts +0 -642
  101. package/dist/index-PdX4VnPA.d.ts +0 -423
  102. package/dist/insight-report-DY4nDW9Q.d.ts +0 -310
  103. package/dist/integrity-DqlBiLyK.d.ts +0 -81
  104. package/dist/judge-calibration-7C-IDmKr.d.ts +0 -145
  105. package/dist/kind-factory-ClZmO25A.d.ts +0 -171
  106. package/dist/llm-client-qoDd18Qz.d.ts +0 -289
  107. package/dist/multi-layer-verifier-BsqKuLyN.d.ts +0 -150
  108. package/dist/off-policy-DiwuKKg7.d.ts +0 -132
  109. package/dist/outcome-store-rnXLEqSn.d.ts +0 -63
  110. package/dist/policy-edit-wG9uFEFm.d.ts +0 -455
  111. package/dist/pre-registration-BWQhJ3vz.d.ts +0 -761
  112. package/dist/provenance-DpjwyseI.d.ts +0 -541
  113. package/dist/query-CF7PG61p.d.ts +0 -35
  114. package/dist/raw-provider-sink-C46HDghv.d.ts +0 -132
  115. package/dist/release-report-C8G2i5Xi.d.ts +0 -236
  116. package/dist/researcher-C8XyxQsu.d.ts +0 -387
  117. package/dist/rubric-predictive-validity-p49lLVrE.d.ts +0 -105
  118. package/dist/run-record-BDH49H2E.d.ts +0 -360
  119. package/dist/runtime-trajectory-DGBIUt4B.d.ts +0 -49
  120. package/dist/schema-B3Q3l9Z_.d.ts +0 -201
  121. package/dist/semantic-concept-judge-CXnPEJbf.d.ts +0 -723
  122. package/dist/sequential-5iSVfzl2.d.ts +0 -139
  123. package/dist/series-convergence-D5OWMBg6.d.ts +0 -33
  124. package/dist/statistics-KUnG73jH.d.ts +0 -494
  125. package/dist/storage-DrX3v_5B.d.ts +0 -50
  126. package/dist/store-C1YxJDEK.d.ts +0 -248
  127. package/dist/store-DGqD0Pyo.d.ts +0 -116
  128. package/dist/summary-report-C5bKFfm-.d.ts +0 -445
  129. package/dist/test-graded-scenario-B0ybnPY7.d.ts +0 -166
  130. package/dist/types-BSw1rOUB.d.ts +0 -634
  131. package/dist/types-BUxNaJ8c.d.ts +0 -108
  132. package/dist/types-BkfcQnxV.d.ts +0 -313
  133. package/dist/verdict-C9MlYujm.d.ts +0 -35
  134. /package/dist/{chunk-VF3XSYTI.js.map → chunk-33JA4TFA.js.map} +0 -0
  135. /package/dist/{chunk-4JLWXDYA.js.map → chunk-3EHHMC6E.js.map} +0 -0
  136. /package/dist/{chunk-CCZIVI3F.js.map → chunk-BFW56GTT.js.map} +0 -0
  137. /package/dist/{chunk-HZHNRYHK.js.map → chunk-K6N6XJJX.js.map} +0 -0
  138. /package/dist/{chunk-MGEHEHSN.js.map → chunk-OIIMMLRB.js.map} +0 -0
  139. /package/dist/{chunk-DXZRATT5.js.map → chunk-OYZAPX5G.js.map} +0 -0
  140. /package/dist/{chunk-JSJZ4PJ6.js.map → chunk-Q442S5AS.js.map} +0 -0
  141. /package/dist/{chunk-ODVOOEWQ.js.map → chunk-QBRSJK47.js.map} +0 -0
  142. /package/dist/{chunk-GQCZRZ7L.js.map → chunk-U5CHZ5M3.js.map} +0 -0
  143. /package/dist/{chunk-HQPHZGL6.js.map → chunk-YLMUS4MM.js.map} +0 -0
@@ -1,132 +0,0 @@
1
- /**
2
- * RawProviderSink — first-class persistence for the actual HTTP-level
3
- * request/response bodies of every LLM provider call.
4
- *
5
- * Why this is a separate sink from the structured `LlmSpan`:
6
- *
7
- * - `LlmSpan` records the *intent* — model name, messages, output text,
8
- * usage. It's what dashboards read; it's NOT enough for forensics.
9
- * - When a downstream consumer reports "the verifier used the wrong route"
10
- * or "tokens look right but reasoning was missing," the only way to
11
- * answer is the raw HTTP body. Span fields can lie (a proxy can echo
12
- * a different `model` value than what actually answered); the raw
13
- * response is ground truth.
14
- *
15
- * Default behaviour: opt-in. Pass `rawSink` to `LlmClientOptions` (or the
16
- * matrix runner / BuilderSession sets it up automatically) and every
17
- * request, response, and error is recorded — including retries, with the
18
- * attempt index attached so a flaky call's full event chain is recoverable.
19
- *
20
- * Redaction is enforced at sink time. The default redactor strips
21
- * `Authorization`, `X-Api-Key`, `X-Auth-Token`, `Cookie` headers and any
22
- * payload field whose key matches `apiKey | api_key | bearer | password |
23
- * secret | token` (case-insensitive). Override via the sink constructor or
24
- * the per-call `redactor`. The `redactedFields` array on the persisted
25
- * event lets a reviewer see what was stripped without exposing the values.
26
- */
27
- type RawProviderDirection = 'request' | 'response' | 'error';
28
- interface RawProviderEvent {
29
- /** Stable id. Generated by the sink if omitted. */
30
- eventId: string;
31
- /** Trace context populated by `LlmClient` when the call is wrapped in a span. */
32
- runId?: string;
33
- spanId?: string;
34
- /**
35
- * Logical provider name. Free-form so callers can use whatever id matches
36
- * their topology (`'openai'`, `'anthropic'`, `'tangle-router'`, …). When
37
- * omitted, derived from `baseUrl` in `LlmClientOptions`.
38
- */
39
- provider: string;
40
- model: string;
41
- /** Endpoint path, e.g. `'/v1/chat/completions'`. */
42
- endpoint: string;
43
- /** Base URL used for the call (already-normalised — no trailing slash). */
44
- baseUrl: string;
45
- /** 0-indexed retry attempt. The first attempt is 0; a retried call gets 1, 2, … */
46
- attemptIndex: number;
47
- direction: RawProviderDirection;
48
- /** Unix ms. */
49
- timestamp: number;
50
- /** Wall-clock duration of the call leg. Set on `response` and `error` events; null on `request`. */
51
- durationMs?: number;
52
- statusCode?: number;
53
- requestHeaders?: Record<string, string>;
54
- requestBody?: unknown;
55
- responseHeaders?: Record<string, string>;
56
- responseBody?: unknown;
57
- /** Set on `direction: 'error'` events. */
58
- errorMessage?: string;
59
- /** Field paths the redactor stripped from this event ('header:Authorization', 'body.apiKey', …). */
60
- redactedFields: string[];
61
- }
62
- interface RawProviderSinkFilter {
63
- runId?: string;
64
- spanId?: string;
65
- direction?: RawProviderDirection;
66
- attemptIndex?: number;
67
- }
68
- interface RawProviderSink {
69
- record(event: RawProviderEvent): Promise<void>;
70
- /** Optional listing — implementations that durably persist (file, db) should support this. */
71
- list?(filter?: RawProviderSinkFilter): Promise<RawProviderEvent[]>;
72
- /** Optional teardown for backed implementations. */
73
- close?(): Promise<void>;
74
- }
75
- type ProviderRedactor = (event: RawProviderEvent) => RawProviderEvent;
76
- /**
77
- * Default redactor — strips well-known auth headers and any body field whose
78
- * key matches the credential pattern. Records every redacted path on
79
- * `event.redactedFields` so a downstream reviewer can see what was removed.
80
- */
81
- declare function defaultProviderRedactor(event: RawProviderEvent): RawProviderEvent;
82
- interface InMemoryRawProviderSinkOptions {
83
- redactor?: ProviderRedactor;
84
- }
85
- declare class InMemoryRawProviderSink implements RawProviderSink {
86
- private events;
87
- private redactor;
88
- constructor(opts?: InMemoryRawProviderSinkOptions);
89
- record(event: RawProviderEvent): Promise<void>;
90
- list(filter?: RawProviderSinkFilter): Promise<RawProviderEvent[]>;
91
- size(): number;
92
- }
93
- declare class NoopRawProviderSink implements RawProviderSink {
94
- record(): Promise<void>;
95
- /**
96
- * Returns an empty array. Implemented so `assertRunCaptured` does not
97
- * trip the `no_raw_sink` issue when a caller explicitly opts out of
98
- * capture by passing this sink — opt-out is a deliberate choice, not a
99
- * misconfiguration.
100
- */
101
- list(): Promise<RawProviderEvent[]>;
102
- }
103
- interface FileSystemRawProviderSinkOptions {
104
- /** Directory the NDJSON file is written into. Created if missing. */
105
- dir: string;
106
- /** File name; default `'raw-provider-events.ndjson'`. */
107
- fileName?: string;
108
- /** Bytes after which the writer rolls over to a new file (default 32 MiB). */
109
- rollAtBytes?: number;
110
- redactor?: ProviderRedactor;
111
- }
112
- declare class FileSystemRawProviderSink implements RawProviderSink {
113
- private dir;
114
- private fileName;
115
- private rollAtBytes;
116
- private redactor;
117
- private bytesWritten;
118
- private rollIndex;
119
- private initPromise;
120
- constructor(opts: FileSystemRawProviderSinkOptions);
121
- private ensureInit;
122
- private currentPath;
123
- record(event: RawProviderEvent): Promise<void>;
124
- list(filter?: RawProviderSinkFilter): Promise<RawProviderEvent[]>;
125
- }
126
- /**
127
- * Best-effort provider id from a base URL. Falls back to the URL host when
128
- * none of the well-known patterns match.
129
- */
130
- declare function providerFromBaseUrl(baseUrl: string): string;
131
-
132
- export { FileSystemRawProviderSink as F, InMemoryRawProviderSink as I, NoopRawProviderSink as N, type ProviderRedactor as P, type RawProviderSink as R, type FileSystemRawProviderSinkOptions as a, type InMemoryRawProviderSinkOptions as b, type RawProviderDirection as c, type RawProviderEvent as d, type RawProviderSinkFilter as e, defaultProviderRedactor as f, providerFromBaseUrl as p };
@@ -1,236 +0,0 @@
1
- import { a as DatasetSplit, b as DatasetManifest, D as DatasetScenario } from './dataset-NENEzRgk.js';
2
- import { m as GateDecision } from './summary-report-C5bKFfm-.js';
3
- import { R as RunRecord, a as RunSplitTag } from './run-record-BDH49H2E.js';
4
-
5
- /**
6
- * Release confidence gate.
7
- *
8
- * This is the production-facing composition layer over the lower-level
9
- * primitives:
10
- * - Dataset manifests prove corpus/version coverage.
11
- * - RunRecord rows prove reproducible search/holdout outcomes.
12
- * - Multi-shot trace evidence carries turn counts and ASI diagnostics.
13
- * - HeldOutGate decisions remain the paired promotion authority.
14
- *
15
- * The gate is intentionally pure and conservative. Missing declared evidence
16
- * fails closed instead of being treated as a neutral zero.
17
- */
18
-
19
- /** Severity of an actionable finding attached to a run/trace. */
20
- type AsiSeverity = 'info' | 'warning' | 'error' | 'critical';
21
- /** Actionable side-info — a diagnosed finding the loop can act on. */
22
- interface ActionableSideInfo {
23
- /** Stable expectation/check id when available. */
24
- expectationId?: string;
25
- /** Human-readable diagnosis of what happened. */
26
- message: string;
27
- severity?: AsiSeverity;
28
- /** Concrete trace excerpt, file path, tool call, screenshot id, etc. */
29
- evidence?: string;
30
- /** Prompt/tool/context surface likely responsible. */
31
- responsibleSurface?: string;
32
- /** Suggested fix in natural language. */
33
- suggestion?: string;
34
- /** Whether this expectation was satisfied. Defaults to false for ASI rows. */
35
- matched?: boolean;
36
- metadata?: Record<string, unknown>;
37
- }
38
- type ReleaseConfidenceStatus = 'pass' | 'warn' | 'fail';
39
- type ReleaseConfidenceAxisName = 'corpus' | 'quality' | 'generalization' | 'diagnostics' | 'efficiency';
40
- interface ReleaseTraceEvidence {
41
- scenarioId: string;
42
- candidateId?: string;
43
- split?: RunSplitTag;
44
- score?: number;
45
- ok?: boolean;
46
- turnCount?: number;
47
- costUsd?: number;
48
- durationMs?: number;
49
- failureMode?: string;
50
- asi?: ActionableSideInfo[];
51
- metadata?: Record<string, unknown>;
52
- }
53
- interface ReleaseConfidenceThresholds {
54
- /** Require a Dataset manifest or explicit scenarios. Default true. */
55
- requireCorpus?: boolean;
56
- minScenarioCount?: number;
57
- minSearchRuns?: number;
58
- minHoldoutRuns?: number;
59
- /** Require at least one holdout scenario/run. Default true. */
60
- requireHoldout?: boolean;
61
- minPassRate?: number;
62
- minMeanScore?: number;
63
- /** Search mean may exceed holdout mean by at most this much. */
64
- maxOverfitGap?: number;
65
- maxMeanCostUsd?: number;
66
- maxP95WallMs?: number;
67
- /** Low-score/failed rows must carry ASI. Default true. */
68
- requireAsiForFailures?: boolean;
69
- /** Score below this is considered a failure for ASI coverage. Default 0.5. */
70
- failureScoreThreshold?: number;
71
- }
72
- interface ReleaseConfidenceInput {
73
- target: string;
74
- candidateId?: string;
75
- baselineId?: string;
76
- dataset?: DatasetManifest;
77
- scenarios?: readonly DatasetScenario[];
78
- runs?: readonly RunRecord[];
79
- traces?: readonly ReleaseTraceEvidence[];
80
- gateDecision?: GateDecision | null;
81
- thresholds?: ReleaseConfidenceThresholds;
82
- }
83
- interface ReleaseConfidenceAxis {
84
- name: ReleaseConfidenceAxisName;
85
- status: ReleaseConfidenceStatus;
86
- score: number;
87
- detail: string;
88
- }
89
- interface ReleaseConfidenceIssue {
90
- axis: ReleaseConfidenceAxisName;
91
- severity: 'critical' | 'warning';
92
- code: string;
93
- detail: string;
94
- }
95
- interface ReleaseConfidenceMetrics {
96
- scenarioCount: number;
97
- searchRuns: number;
98
- holdoutRuns: number;
99
- passRate: number;
100
- meanScore: number;
101
- searchMeanScore: number;
102
- holdoutMeanScore: number;
103
- overfitGap: number;
104
- meanCostUsd: number;
105
- p95WallMs: number;
106
- failedRows: number;
107
- failuresWithAsi: number;
108
- singleShotTraces: number;
109
- multiShotTraces: number;
110
- splitCounts: Record<DatasetSplit, number>;
111
- domainCounts: Record<string, number>;
112
- failureModeCounts: Record<string, number>;
113
- responsibleSurfaceCounts: Record<string, number>;
114
- }
115
- interface ReleaseConfidenceScorecard {
116
- target: string;
117
- candidateId: string | null;
118
- baselineId: string | null;
119
- status: ReleaseConfidenceStatus;
120
- promote: boolean;
121
- axes: ReleaseConfidenceAxis[];
122
- issues: ReleaseConfidenceIssue[];
123
- metrics: ReleaseConfidenceMetrics;
124
- dataset: DatasetManifest | null;
125
- gateDecision: GateDecision | null;
126
- summary: string;
127
- }
128
- declare function evaluateReleaseConfidence(input: ReleaseConfidenceInput): ReleaseConfidenceScorecard;
129
- declare function assertReleaseConfidence(input: ReleaseConfidenceInput): ReleaseConfidenceScorecard;
130
-
131
- /**
132
- * Bootstrap-CI promotion gate.
133
- *
134
- * In any iterative-improvement loop (GEPA, prompt evolution, dataset
135
- * curation), the question is "did this generation actually improve, or are
136
- * we celebrating noise?". With small N and noisy outcomes, point-estimate
137
- * deltas lie. Bootstrap confidence intervals tell the operator whether the
138
- * delta is real before code or prompts get promoted.
139
- *
140
- * This module is pure functions — no I/O, no model calls. Easy to unit-test
141
- * and to compose into any verdict gate.
142
- *
143
- * Default gate:
144
- * - Bootstrap mean baseline vs candidate (1k resamples).
145
- * - Compute the delta distribution; pass if the lower CI bound > 0.
146
- * - Tunable confidence (default 95%) and resample count.
147
- *
148
- * Verdict semantics intentionally match the existing `experiments.jsonl`
149
- * vocabulary:
150
- * - ADVANCE: candidate's CI lower bound > baseline mean (real win)
151
- * - KEEP: overlap, but candidate point estimate >= baseline (neutral)
152
- * - REVERT: candidate's CI upper bound < baseline mean (real regression)
153
- * - INCONCLUSIVE: not enough samples or CI straddles zero with no signal
154
- */
155
- type Verdict = 'ADVANCE' | 'KEEP' | 'REVERT' | 'INCONCLUSIVE';
156
- interface BootstrapResult {
157
- baselineMean: number;
158
- candidateMean: number;
159
- /** candidateMean - baselineMean, point estimate. */
160
- delta: number;
161
- /** Lower bound of the (1 - alpha) CI on the delta. */
162
- ciLower: number;
163
- /** Upper bound of the (1 - alpha) CI on the delta. */
164
- ciUpper: number;
165
- /** Number of bootstrap resamples used. */
166
- iterations: number;
167
- alpha: number;
168
- verdict: Verdict;
169
- }
170
- interface BootstrapOptions {
171
- /** Confidence level alpha (default 0.05 → 95% CI). */
172
- alpha?: number;
173
- /** Number of resamples (default 1000). */
174
- iterations?: number;
175
- /**
176
- * Minimum total samples (baseline + candidate) below which we always
177
- * return INCONCLUSIVE — bootstrap with too few samples is meaningless.
178
- * Default 6 (combined).
179
- */
180
- minTotalSamples?: number;
181
- /** RNG seed for reproducibility. Default: Math.random. */
182
- seed?: number;
183
- }
184
- /**
185
- * Compute the bootstrap CI on (candidateMean - baselineMean) and a verdict.
186
- *
187
- * Uses simple percentile bootstrap on the difference of resampled means.
188
- * That's the standard non-parametric primitive — no distributional
189
- * assumptions, robust to skew, easy to reason about.
190
- */
191
- declare function bootstrapCi(baseline: number[], candidate: number[], options?: BootstrapOptions): BootstrapResult;
192
- /**
193
- * Judge-replay promotion gate.
194
- *
195
- * The cheap inner-loop judge that drives an evolution run is by definition
196
- * fast and noisy. When you're about to promote a winning variant to the
197
- * canonical default, you want a STRONGER judge (a more expensive model, a
198
- * human grader, a separately-trained reward model) to confirm the win
199
- * generalises beyond the inner loop.
200
- *
201
- * This helper takes raw winner + baseline outputs, scores both through the
202
- * stronger judge, and applies `bootstrapCi`. ADVANCE means the stronger
203
- * judge agrees the winner is real with the configured confidence. Doesn't
204
- * matter what shape your "output" is — pass a string, an object, anything
205
- * the judge can read.
206
- */
207
- interface JudgeReplayGateArgs<TOutput> {
208
- baselineOutputs: TOutput[];
209
- candidateOutputs: TOutput[];
210
- /** Stronger judge — async to allow LLM calls. Return a 0..N scalar score. */
211
- judge: (output: TOutput) => Promise<number> | number;
212
- alpha?: number;
213
- iterations?: number;
214
- /** RNG seed for reproducibility. */
215
- seed?: number;
216
- /** Maximum concurrent judge calls. Default 4. */
217
- judgeConcurrency?: number;
218
- }
219
- /**
220
- * Confirm a candidate's win with a stronger judge: score baseline and candidate outputs independently, then bootstrap a CI to verify the lift generalises beyond the inner loop.
221
- */
222
- declare function judgeReplayGate<TOutput>(args: JudgeReplayGateArgs<TOutput>): Promise<BootstrapResult & {
223
- baselineSamples: number;
224
- candidateSamples: number;
225
- }>;
226
-
227
- interface RenderReleaseReportOptions {
228
- title?: string;
229
- runs?: readonly RunRecord[];
230
- comparator?: string;
231
- traceAnalystFindings?: readonly string[];
232
- nextActions?: readonly string[];
233
- }
234
- declare function renderReleaseReport(scorecard: ReleaseConfidenceScorecard, options?: RenderReleaseReportOptions): string;
235
-
236
- export { type ActionableSideInfo as A, type BootstrapOptions as B, type JudgeReplayGateArgs as J, type ReleaseConfidenceAxis as R, type Verdict as V, type BootstrapResult as a, type ReleaseConfidenceAxisName as b, type ReleaseConfidenceInput as c, type ReleaseConfidenceIssue as d, type ReleaseConfidenceMetrics as e, type ReleaseConfidenceScorecard as f, type ReleaseConfidenceStatus as g, type ReleaseConfidenceThresholds as h, type ReleaseTraceEvidence as i, type RenderReleaseReportOptions as j, assertReleaseConfidence as k, bootstrapCi as l, evaluateReleaseConfidence as m, judgeReplayGate as n, type AsiSeverity as o, renderReleaseReport as r };