@tangle-network/agent-eval 0.117.1 → 0.118.1
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +30 -0
- package/dist/analyst/index.d.ts +2772 -21
- package/dist/analyst/index.js +7 -6
- package/dist/analyst/index.js.map +1 -1
- package/dist/belief-state/index.d.ts +706 -10
- package/dist/belief-state/index.js +2 -1
- package/dist/belief-state/index.js.map +1 -1
- package/dist/benchmarks/index.d.ts +958 -14
- package/dist/benchmarks/index.js +12 -10
- package/dist/builder-eval/index.d.ts +449 -4
- package/dist/builder-eval/index.js +4 -3
- package/dist/builder-eval/index.js.map +1 -1
- package/dist/campaign/index.d.ts +4275 -73
- package/dist/campaign/index.js +12 -10
- package/dist/{chunk-VF3XSYTI.js → chunk-33JA4TFA.js} +6 -6
- package/dist/{chunk-4JLWXDYA.js → chunk-3EHHMC6E.js} +2 -2
- package/dist/{chunk-CCZIVI3F.js → chunk-BFW56GTT.js} +2 -2
- package/dist/{chunk-YZPO4UHR.js → chunk-FTUMG2U7.js} +124 -149
- package/dist/chunk-FTUMG2U7.js.map +1 -0
- package/dist/{chunk-E4BUPP7Z.js → chunk-HKUCJ437.js} +38 -66
- package/dist/chunk-HKUCJ437.js.map +1 -0
- package/dist/{chunk-HZHNRYHK.js → chunk-K6N6XJJX.js} +2 -2
- package/dist/chunk-KSDQVPLR.js +286 -0
- package/dist/chunk-KSDQVPLR.js.map +1 -0
- package/dist/chunk-MA6HLL3S.js +65 -0
- package/dist/chunk-MA6HLL3S.js.map +1 -0
- package/dist/{chunk-MGEHEHSN.js → chunk-OIIMMLRB.js} +11 -11
- package/dist/{chunk-DXZRATT5.js → chunk-OYZAPX5G.js} +3 -3
- package/dist/chunk-PXE2VKMX.js +140 -0
- package/dist/chunk-PXE2VKMX.js.map +1 -0
- package/dist/{chunk-JSJZ4PJ6.js → chunk-Q442S5AS.js} +17 -17
- package/dist/{chunk-ODVOOEWQ.js → chunk-QBRSJK47.js} +2 -2
- package/dist/{chunk-S2F4J57L.js → chunk-QKEGNI5B.js} +77 -32
- package/dist/chunk-QKEGNI5B.js.map +1 -0
- package/dist/{chunk-5UF54T55.js → chunk-S3UZOQ5Y.js} +34 -7
- package/dist/chunk-S3UZOQ5Y.js.map +1 -0
- package/dist/{chunk-FQNLDL4D.js → chunk-SVH2ANFD.js} +136 -4
- package/dist/chunk-SVH2ANFD.js.map +1 -0
- package/dist/{chunk-GQCZRZ7L.js → chunk-U5CHZ5M3.js} +9 -9
- package/dist/{chunk-TVVP3ZZQ.js → chunk-VQMK5FMP.js} +2 -1
- package/dist/{chunk-TVVP3ZZQ.js.map → chunk-VQMK5FMP.js.map} +1 -1
- package/dist/chunk-WDHBCA3M.js +31 -0
- package/dist/chunk-WDHBCA3M.js.map +1 -0
- package/dist/{chunk-HQPHZGL6.js → chunk-YLMUS4MM.js} +9 -9
- package/dist/{chunk-LQUTGLOZ.js → chunk-ZET2UAYW.js} +15 -65
- package/dist/chunk-ZET2UAYW.js.map +1 -0
- package/dist/contract/index.d.ts +4012 -38
- package/dist/contract/index.js +56 -29
- package/dist/contract/index.js.map +1 -1
- package/dist/control.d.ts +1013 -9
- package/dist/control.js +4 -3
- package/dist/fuzz.d.ts +194 -4
- package/dist/fuzz.js +3 -3
- package/dist/hosted/index.d.ts +498 -17
- package/dist/index.d.ts +10992 -1288
- package/dist/index.js +101 -80
- package/dist/index.js.map +1 -1
- package/dist/matrix/index.d.ts +139 -4
- package/dist/meta-eval/index.d.ts +862 -15
- package/dist/meta-eval/index.js +2 -1
- package/dist/meta-eval/index.js.map +1 -1
- package/dist/multishot/index.d.ts +214 -14
- package/dist/openapi.json +1 -1
- package/dist/pipelines/index.d.ts +392 -7
- package/dist/pipelines/index.js +5 -3
- package/dist/pipelines/index.js.map +1 -1
- package/dist/reporting.d.ts +1277 -17
- package/dist/rl.d.ts +2359 -28
- package/dist/rl.js +6 -5
- package/dist/rl.js.map +1 -1
- package/dist/storyboard/index.d.ts +86 -1
- package/dist/trace-attributes.d.ts +16 -0
- package/dist/trace-attributes.js +32 -0
- package/dist/trace-attributes.js.map +1 -0
- package/dist/traces.d.ts +1978 -697
- package/dist/traces.js +53 -32
- package/dist/wire/index.d.ts +655 -9
- package/docs/insight-report.md +44 -0
- package/package.json +9 -3
- package/dist/adversarial-B7loGVVX.d.ts +0 -19
- package/dist/analyst-C8HHvfJp.d.ts +0 -88
- package/dist/analyze-runs--2x39HZ7.d.ts +0 -81
- package/dist/baseline-DKq3gJpP.d.ts +0 -141
- package/dist/calibration-C8MTS7cw.d.ts +0 -101
- package/dist/chunk-5UF54T55.js.map +0 -1
- package/dist/chunk-E4BUPP7Z.js.map +0 -1
- package/dist/chunk-FQNLDL4D.js.map +0 -1
- package/dist/chunk-LQUTGLOZ.js.map +0 -1
- package/dist/chunk-S2F4J57L.js.map +0 -1
- package/dist/chunk-YZPO4UHR.js.map +0 -1
- package/dist/code-agent-session-CjZsVd19.d.ts +0 -87
- package/dist/control-6vuGfmDH.d.ts +0 -258
- package/dist/cost-ledger-DWy3XdJc.d.ts +0 -183
- package/dist/dataset-NENEzRgk.d.ts +0 -115
- package/dist/default-registry-DaK8b3fv.d.ts +0 -155
- package/dist/emitter-CjD7vUwv.d.ts +0 -122
- package/dist/errors-oeQrLqXC.d.ts +0 -74
- package/dist/failure-cluster-DOAcSJ87.d.ts +0 -76
- package/dist/feedback-trajectory-BUnM58xL.d.ts +0 -348
- package/dist/gepa-eESocoDi.d.ts +0 -642
- package/dist/index-PdX4VnPA.d.ts +0 -423
- package/dist/insight-report-DY4nDW9Q.d.ts +0 -310
- package/dist/integrity-DqlBiLyK.d.ts +0 -81
- package/dist/judge-calibration-7C-IDmKr.d.ts +0 -145
- package/dist/kind-factory-ClZmO25A.d.ts +0 -171
- package/dist/llm-client-qoDd18Qz.d.ts +0 -289
- package/dist/multi-layer-verifier-BsqKuLyN.d.ts +0 -150
- package/dist/off-policy-DiwuKKg7.d.ts +0 -132
- package/dist/outcome-store-rnXLEqSn.d.ts +0 -63
- package/dist/policy-edit-wG9uFEFm.d.ts +0 -455
- package/dist/pre-registration-BWQhJ3vz.d.ts +0 -761
- package/dist/provenance-DpjwyseI.d.ts +0 -541
- package/dist/query-CF7PG61p.d.ts +0 -35
- package/dist/raw-provider-sink-C46HDghv.d.ts +0 -132
- package/dist/release-report-C8G2i5Xi.d.ts +0 -236
- package/dist/researcher-C8XyxQsu.d.ts +0 -387
- package/dist/rubric-predictive-validity-p49lLVrE.d.ts +0 -105
- package/dist/run-record-BDH49H2E.d.ts +0 -360
- package/dist/runtime-trajectory-DGBIUt4B.d.ts +0 -49
- package/dist/schema-B3Q3l9Z_.d.ts +0 -201
- package/dist/semantic-concept-judge-CXnPEJbf.d.ts +0 -723
- package/dist/sequential-5iSVfzl2.d.ts +0 -139
- package/dist/series-convergence-D5OWMBg6.d.ts +0 -33
- package/dist/statistics-KUnG73jH.d.ts +0 -494
- package/dist/storage-DrX3v_5B.d.ts +0 -50
- package/dist/store-C1YxJDEK.d.ts +0 -248
- package/dist/store-DGqD0Pyo.d.ts +0 -116
- package/dist/summary-report-C5bKFfm-.d.ts +0 -445
- package/dist/test-graded-scenario-B0ybnPY7.d.ts +0 -166
- package/dist/types-BSw1rOUB.d.ts +0 -634
- package/dist/types-BUxNaJ8c.d.ts +0 -108
- package/dist/types-BkfcQnxV.d.ts +0 -313
- package/dist/verdict-C9MlYujm.d.ts +0 -35
- /package/dist/{chunk-VF3XSYTI.js.map → chunk-33JA4TFA.js.map} +0 -0
- /package/dist/{chunk-4JLWXDYA.js.map → chunk-3EHHMC6E.js.map} +0 -0
- /package/dist/{chunk-CCZIVI3F.js.map → chunk-BFW56GTT.js.map} +0 -0
- /package/dist/{chunk-HZHNRYHK.js.map → chunk-K6N6XJJX.js.map} +0 -0
- /package/dist/{chunk-MGEHEHSN.js.map → chunk-OIIMMLRB.js.map} +0 -0
- /package/dist/{chunk-DXZRATT5.js.map → chunk-OYZAPX5G.js.map} +0 -0
- /package/dist/{chunk-JSJZ4PJ6.js.map → chunk-Q442S5AS.js.map} +0 -0
- /package/dist/{chunk-ODVOOEWQ.js.map → chunk-QBRSJK47.js.map} +0 -0
- /package/dist/{chunk-GQCZRZ7L.js.map → chunk-U5CHZ5M3.js.map} +0 -0
- /package/dist/{chunk-HQPHZGL6.js.map → chunk-YLMUS4MM.js.map} +0 -0
|
@@ -1,132 +0,0 @@
|
|
|
1
|
-
/**
|
|
2
|
-
* RawProviderSink — first-class persistence for the actual HTTP-level
|
|
3
|
-
* request/response bodies of every LLM provider call.
|
|
4
|
-
*
|
|
5
|
-
* Why this is a separate sink from the structured `LlmSpan`:
|
|
6
|
-
*
|
|
7
|
-
* - `LlmSpan` records the *intent* — model name, messages, output text,
|
|
8
|
-
* usage. It's what dashboards read; it's NOT enough for forensics.
|
|
9
|
-
* - When a downstream consumer reports "the verifier used the wrong route"
|
|
10
|
-
* or "tokens look right but reasoning was missing," the only way to
|
|
11
|
-
* answer is the raw HTTP body. Span fields can lie (a proxy can echo
|
|
12
|
-
* a different `model` value than what actually answered); the raw
|
|
13
|
-
* response is ground truth.
|
|
14
|
-
*
|
|
15
|
-
* Default behaviour: opt-in. Pass `rawSink` to `LlmClientOptions` (or the
|
|
16
|
-
* matrix runner / BuilderSession sets it up automatically) and every
|
|
17
|
-
* request, response, and error is recorded — including retries, with the
|
|
18
|
-
* attempt index attached so a flaky call's full event chain is recoverable.
|
|
19
|
-
*
|
|
20
|
-
* Redaction is enforced at sink time. The default redactor strips
|
|
21
|
-
* `Authorization`, `X-Api-Key`, `X-Auth-Token`, `Cookie` headers and any
|
|
22
|
-
* payload field whose key matches `apiKey | api_key | bearer | password |
|
|
23
|
-
* secret | token` (case-insensitive). Override via the sink constructor or
|
|
24
|
-
* the per-call `redactor`. The `redactedFields` array on the persisted
|
|
25
|
-
* event lets a reviewer see what was stripped without exposing the values.
|
|
26
|
-
*/
|
|
27
|
-
type RawProviderDirection = 'request' | 'response' | 'error';
|
|
28
|
-
interface RawProviderEvent {
|
|
29
|
-
/** Stable id. Generated by the sink if omitted. */
|
|
30
|
-
eventId: string;
|
|
31
|
-
/** Trace context populated by `LlmClient` when the call is wrapped in a span. */
|
|
32
|
-
runId?: string;
|
|
33
|
-
spanId?: string;
|
|
34
|
-
/**
|
|
35
|
-
* Logical provider name. Free-form so callers can use whatever id matches
|
|
36
|
-
* their topology (`'openai'`, `'anthropic'`, `'tangle-router'`, …). When
|
|
37
|
-
* omitted, derived from `baseUrl` in `LlmClientOptions`.
|
|
38
|
-
*/
|
|
39
|
-
provider: string;
|
|
40
|
-
model: string;
|
|
41
|
-
/** Endpoint path, e.g. `'/v1/chat/completions'`. */
|
|
42
|
-
endpoint: string;
|
|
43
|
-
/** Base URL used for the call (already-normalised — no trailing slash). */
|
|
44
|
-
baseUrl: string;
|
|
45
|
-
/** 0-indexed retry attempt. The first attempt is 0; a retried call gets 1, 2, … */
|
|
46
|
-
attemptIndex: number;
|
|
47
|
-
direction: RawProviderDirection;
|
|
48
|
-
/** Unix ms. */
|
|
49
|
-
timestamp: number;
|
|
50
|
-
/** Wall-clock duration of the call leg. Set on `response` and `error` events; null on `request`. */
|
|
51
|
-
durationMs?: number;
|
|
52
|
-
statusCode?: number;
|
|
53
|
-
requestHeaders?: Record<string, string>;
|
|
54
|
-
requestBody?: unknown;
|
|
55
|
-
responseHeaders?: Record<string, string>;
|
|
56
|
-
responseBody?: unknown;
|
|
57
|
-
/** Set on `direction: 'error'` events. */
|
|
58
|
-
errorMessage?: string;
|
|
59
|
-
/** Field paths the redactor stripped from this event ('header:Authorization', 'body.apiKey', …). */
|
|
60
|
-
redactedFields: string[];
|
|
61
|
-
}
|
|
62
|
-
interface RawProviderSinkFilter {
|
|
63
|
-
runId?: string;
|
|
64
|
-
spanId?: string;
|
|
65
|
-
direction?: RawProviderDirection;
|
|
66
|
-
attemptIndex?: number;
|
|
67
|
-
}
|
|
68
|
-
interface RawProviderSink {
|
|
69
|
-
record(event: RawProviderEvent): Promise<void>;
|
|
70
|
-
/** Optional listing — implementations that durably persist (file, db) should support this. */
|
|
71
|
-
list?(filter?: RawProviderSinkFilter): Promise<RawProviderEvent[]>;
|
|
72
|
-
/** Optional teardown for backed implementations. */
|
|
73
|
-
close?(): Promise<void>;
|
|
74
|
-
}
|
|
75
|
-
type ProviderRedactor = (event: RawProviderEvent) => RawProviderEvent;
|
|
76
|
-
/**
|
|
77
|
-
* Default redactor — strips well-known auth headers and any body field whose
|
|
78
|
-
* key matches the credential pattern. Records every redacted path on
|
|
79
|
-
* `event.redactedFields` so a downstream reviewer can see what was removed.
|
|
80
|
-
*/
|
|
81
|
-
declare function defaultProviderRedactor(event: RawProviderEvent): RawProviderEvent;
|
|
82
|
-
interface InMemoryRawProviderSinkOptions {
|
|
83
|
-
redactor?: ProviderRedactor;
|
|
84
|
-
}
|
|
85
|
-
declare class InMemoryRawProviderSink implements RawProviderSink {
|
|
86
|
-
private events;
|
|
87
|
-
private redactor;
|
|
88
|
-
constructor(opts?: InMemoryRawProviderSinkOptions);
|
|
89
|
-
record(event: RawProviderEvent): Promise<void>;
|
|
90
|
-
list(filter?: RawProviderSinkFilter): Promise<RawProviderEvent[]>;
|
|
91
|
-
size(): number;
|
|
92
|
-
}
|
|
93
|
-
declare class NoopRawProviderSink implements RawProviderSink {
|
|
94
|
-
record(): Promise<void>;
|
|
95
|
-
/**
|
|
96
|
-
* Returns an empty array. Implemented so `assertRunCaptured` does not
|
|
97
|
-
* trip the `no_raw_sink` issue when a caller explicitly opts out of
|
|
98
|
-
* capture by passing this sink — opt-out is a deliberate choice, not a
|
|
99
|
-
* misconfiguration.
|
|
100
|
-
*/
|
|
101
|
-
list(): Promise<RawProviderEvent[]>;
|
|
102
|
-
}
|
|
103
|
-
interface FileSystemRawProviderSinkOptions {
|
|
104
|
-
/** Directory the NDJSON file is written into. Created if missing. */
|
|
105
|
-
dir: string;
|
|
106
|
-
/** File name; default `'raw-provider-events.ndjson'`. */
|
|
107
|
-
fileName?: string;
|
|
108
|
-
/** Bytes after which the writer rolls over to a new file (default 32 MiB). */
|
|
109
|
-
rollAtBytes?: number;
|
|
110
|
-
redactor?: ProviderRedactor;
|
|
111
|
-
}
|
|
112
|
-
declare class FileSystemRawProviderSink implements RawProviderSink {
|
|
113
|
-
private dir;
|
|
114
|
-
private fileName;
|
|
115
|
-
private rollAtBytes;
|
|
116
|
-
private redactor;
|
|
117
|
-
private bytesWritten;
|
|
118
|
-
private rollIndex;
|
|
119
|
-
private initPromise;
|
|
120
|
-
constructor(opts: FileSystemRawProviderSinkOptions);
|
|
121
|
-
private ensureInit;
|
|
122
|
-
private currentPath;
|
|
123
|
-
record(event: RawProviderEvent): Promise<void>;
|
|
124
|
-
list(filter?: RawProviderSinkFilter): Promise<RawProviderEvent[]>;
|
|
125
|
-
}
|
|
126
|
-
/**
|
|
127
|
-
* Best-effort provider id from a base URL. Falls back to the URL host when
|
|
128
|
-
* none of the well-known patterns match.
|
|
129
|
-
*/
|
|
130
|
-
declare function providerFromBaseUrl(baseUrl: string): string;
|
|
131
|
-
|
|
132
|
-
export { FileSystemRawProviderSink as F, InMemoryRawProviderSink as I, NoopRawProviderSink as N, type ProviderRedactor as P, type RawProviderSink as R, type FileSystemRawProviderSinkOptions as a, type InMemoryRawProviderSinkOptions as b, type RawProviderDirection as c, type RawProviderEvent as d, type RawProviderSinkFilter as e, defaultProviderRedactor as f, providerFromBaseUrl as p };
|
|
@@ -1,236 +0,0 @@
|
|
|
1
|
-
import { a as DatasetSplit, b as DatasetManifest, D as DatasetScenario } from './dataset-NENEzRgk.js';
|
|
2
|
-
import { m as GateDecision } from './summary-report-C5bKFfm-.js';
|
|
3
|
-
import { R as RunRecord, a as RunSplitTag } from './run-record-BDH49H2E.js';
|
|
4
|
-
|
|
5
|
-
/**
|
|
6
|
-
* Release confidence gate.
|
|
7
|
-
*
|
|
8
|
-
* This is the production-facing composition layer over the lower-level
|
|
9
|
-
* primitives:
|
|
10
|
-
* - Dataset manifests prove corpus/version coverage.
|
|
11
|
-
* - RunRecord rows prove reproducible search/holdout outcomes.
|
|
12
|
-
* - Multi-shot trace evidence carries turn counts and ASI diagnostics.
|
|
13
|
-
* - HeldOutGate decisions remain the paired promotion authority.
|
|
14
|
-
*
|
|
15
|
-
* The gate is intentionally pure and conservative. Missing declared evidence
|
|
16
|
-
* fails closed instead of being treated as a neutral zero.
|
|
17
|
-
*/
|
|
18
|
-
|
|
19
|
-
/** Severity of an actionable finding attached to a run/trace. */
|
|
20
|
-
type AsiSeverity = 'info' | 'warning' | 'error' | 'critical';
|
|
21
|
-
/** Actionable side-info — a diagnosed finding the loop can act on. */
|
|
22
|
-
interface ActionableSideInfo {
|
|
23
|
-
/** Stable expectation/check id when available. */
|
|
24
|
-
expectationId?: string;
|
|
25
|
-
/** Human-readable diagnosis of what happened. */
|
|
26
|
-
message: string;
|
|
27
|
-
severity?: AsiSeverity;
|
|
28
|
-
/** Concrete trace excerpt, file path, tool call, screenshot id, etc. */
|
|
29
|
-
evidence?: string;
|
|
30
|
-
/** Prompt/tool/context surface likely responsible. */
|
|
31
|
-
responsibleSurface?: string;
|
|
32
|
-
/** Suggested fix in natural language. */
|
|
33
|
-
suggestion?: string;
|
|
34
|
-
/** Whether this expectation was satisfied. Defaults to false for ASI rows. */
|
|
35
|
-
matched?: boolean;
|
|
36
|
-
metadata?: Record<string, unknown>;
|
|
37
|
-
}
|
|
38
|
-
type ReleaseConfidenceStatus = 'pass' | 'warn' | 'fail';
|
|
39
|
-
type ReleaseConfidenceAxisName = 'corpus' | 'quality' | 'generalization' | 'diagnostics' | 'efficiency';
|
|
40
|
-
interface ReleaseTraceEvidence {
|
|
41
|
-
scenarioId: string;
|
|
42
|
-
candidateId?: string;
|
|
43
|
-
split?: RunSplitTag;
|
|
44
|
-
score?: number;
|
|
45
|
-
ok?: boolean;
|
|
46
|
-
turnCount?: number;
|
|
47
|
-
costUsd?: number;
|
|
48
|
-
durationMs?: number;
|
|
49
|
-
failureMode?: string;
|
|
50
|
-
asi?: ActionableSideInfo[];
|
|
51
|
-
metadata?: Record<string, unknown>;
|
|
52
|
-
}
|
|
53
|
-
interface ReleaseConfidenceThresholds {
|
|
54
|
-
/** Require a Dataset manifest or explicit scenarios. Default true. */
|
|
55
|
-
requireCorpus?: boolean;
|
|
56
|
-
minScenarioCount?: number;
|
|
57
|
-
minSearchRuns?: number;
|
|
58
|
-
minHoldoutRuns?: number;
|
|
59
|
-
/** Require at least one holdout scenario/run. Default true. */
|
|
60
|
-
requireHoldout?: boolean;
|
|
61
|
-
minPassRate?: number;
|
|
62
|
-
minMeanScore?: number;
|
|
63
|
-
/** Search mean may exceed holdout mean by at most this much. */
|
|
64
|
-
maxOverfitGap?: number;
|
|
65
|
-
maxMeanCostUsd?: number;
|
|
66
|
-
maxP95WallMs?: number;
|
|
67
|
-
/** Low-score/failed rows must carry ASI. Default true. */
|
|
68
|
-
requireAsiForFailures?: boolean;
|
|
69
|
-
/** Score below this is considered a failure for ASI coverage. Default 0.5. */
|
|
70
|
-
failureScoreThreshold?: number;
|
|
71
|
-
}
|
|
72
|
-
interface ReleaseConfidenceInput {
|
|
73
|
-
target: string;
|
|
74
|
-
candidateId?: string;
|
|
75
|
-
baselineId?: string;
|
|
76
|
-
dataset?: DatasetManifest;
|
|
77
|
-
scenarios?: readonly DatasetScenario[];
|
|
78
|
-
runs?: readonly RunRecord[];
|
|
79
|
-
traces?: readonly ReleaseTraceEvidence[];
|
|
80
|
-
gateDecision?: GateDecision | null;
|
|
81
|
-
thresholds?: ReleaseConfidenceThresholds;
|
|
82
|
-
}
|
|
83
|
-
interface ReleaseConfidenceAxis {
|
|
84
|
-
name: ReleaseConfidenceAxisName;
|
|
85
|
-
status: ReleaseConfidenceStatus;
|
|
86
|
-
score: number;
|
|
87
|
-
detail: string;
|
|
88
|
-
}
|
|
89
|
-
interface ReleaseConfidenceIssue {
|
|
90
|
-
axis: ReleaseConfidenceAxisName;
|
|
91
|
-
severity: 'critical' | 'warning';
|
|
92
|
-
code: string;
|
|
93
|
-
detail: string;
|
|
94
|
-
}
|
|
95
|
-
interface ReleaseConfidenceMetrics {
|
|
96
|
-
scenarioCount: number;
|
|
97
|
-
searchRuns: number;
|
|
98
|
-
holdoutRuns: number;
|
|
99
|
-
passRate: number;
|
|
100
|
-
meanScore: number;
|
|
101
|
-
searchMeanScore: number;
|
|
102
|
-
holdoutMeanScore: number;
|
|
103
|
-
overfitGap: number;
|
|
104
|
-
meanCostUsd: number;
|
|
105
|
-
p95WallMs: number;
|
|
106
|
-
failedRows: number;
|
|
107
|
-
failuresWithAsi: number;
|
|
108
|
-
singleShotTraces: number;
|
|
109
|
-
multiShotTraces: number;
|
|
110
|
-
splitCounts: Record<DatasetSplit, number>;
|
|
111
|
-
domainCounts: Record<string, number>;
|
|
112
|
-
failureModeCounts: Record<string, number>;
|
|
113
|
-
responsibleSurfaceCounts: Record<string, number>;
|
|
114
|
-
}
|
|
115
|
-
interface ReleaseConfidenceScorecard {
|
|
116
|
-
target: string;
|
|
117
|
-
candidateId: string | null;
|
|
118
|
-
baselineId: string | null;
|
|
119
|
-
status: ReleaseConfidenceStatus;
|
|
120
|
-
promote: boolean;
|
|
121
|
-
axes: ReleaseConfidenceAxis[];
|
|
122
|
-
issues: ReleaseConfidenceIssue[];
|
|
123
|
-
metrics: ReleaseConfidenceMetrics;
|
|
124
|
-
dataset: DatasetManifest | null;
|
|
125
|
-
gateDecision: GateDecision | null;
|
|
126
|
-
summary: string;
|
|
127
|
-
}
|
|
128
|
-
declare function evaluateReleaseConfidence(input: ReleaseConfidenceInput): ReleaseConfidenceScorecard;
|
|
129
|
-
declare function assertReleaseConfidence(input: ReleaseConfidenceInput): ReleaseConfidenceScorecard;
|
|
130
|
-
|
|
131
|
-
/**
|
|
132
|
-
* Bootstrap-CI promotion gate.
|
|
133
|
-
*
|
|
134
|
-
* In any iterative-improvement loop (GEPA, prompt evolution, dataset
|
|
135
|
-
* curation), the question is "did this generation actually improve, or are
|
|
136
|
-
* we celebrating noise?". With small N and noisy outcomes, point-estimate
|
|
137
|
-
* deltas lie. Bootstrap confidence intervals tell the operator whether the
|
|
138
|
-
* delta is real before code or prompts get promoted.
|
|
139
|
-
*
|
|
140
|
-
* This module is pure functions — no I/O, no model calls. Easy to unit-test
|
|
141
|
-
* and to compose into any verdict gate.
|
|
142
|
-
*
|
|
143
|
-
* Default gate:
|
|
144
|
-
* - Bootstrap mean baseline vs candidate (1k resamples).
|
|
145
|
-
* - Compute the delta distribution; pass if the lower CI bound > 0.
|
|
146
|
-
* - Tunable confidence (default 95%) and resample count.
|
|
147
|
-
*
|
|
148
|
-
* Verdict semantics intentionally match the existing `experiments.jsonl`
|
|
149
|
-
* vocabulary:
|
|
150
|
-
* - ADVANCE: candidate's CI lower bound > baseline mean (real win)
|
|
151
|
-
* - KEEP: overlap, but candidate point estimate >= baseline (neutral)
|
|
152
|
-
* - REVERT: candidate's CI upper bound < baseline mean (real regression)
|
|
153
|
-
* - INCONCLUSIVE: not enough samples or CI straddles zero with no signal
|
|
154
|
-
*/
|
|
155
|
-
type Verdict = 'ADVANCE' | 'KEEP' | 'REVERT' | 'INCONCLUSIVE';
|
|
156
|
-
interface BootstrapResult {
|
|
157
|
-
baselineMean: number;
|
|
158
|
-
candidateMean: number;
|
|
159
|
-
/** candidateMean - baselineMean, point estimate. */
|
|
160
|
-
delta: number;
|
|
161
|
-
/** Lower bound of the (1 - alpha) CI on the delta. */
|
|
162
|
-
ciLower: number;
|
|
163
|
-
/** Upper bound of the (1 - alpha) CI on the delta. */
|
|
164
|
-
ciUpper: number;
|
|
165
|
-
/** Number of bootstrap resamples used. */
|
|
166
|
-
iterations: number;
|
|
167
|
-
alpha: number;
|
|
168
|
-
verdict: Verdict;
|
|
169
|
-
}
|
|
170
|
-
interface BootstrapOptions {
|
|
171
|
-
/** Confidence level alpha (default 0.05 → 95% CI). */
|
|
172
|
-
alpha?: number;
|
|
173
|
-
/** Number of resamples (default 1000). */
|
|
174
|
-
iterations?: number;
|
|
175
|
-
/**
|
|
176
|
-
* Minimum total samples (baseline + candidate) below which we always
|
|
177
|
-
* return INCONCLUSIVE — bootstrap with too few samples is meaningless.
|
|
178
|
-
* Default 6 (combined).
|
|
179
|
-
*/
|
|
180
|
-
minTotalSamples?: number;
|
|
181
|
-
/** RNG seed for reproducibility. Default: Math.random. */
|
|
182
|
-
seed?: number;
|
|
183
|
-
}
|
|
184
|
-
/**
|
|
185
|
-
* Compute the bootstrap CI on (candidateMean - baselineMean) and a verdict.
|
|
186
|
-
*
|
|
187
|
-
* Uses simple percentile bootstrap on the difference of resampled means.
|
|
188
|
-
* That's the standard non-parametric primitive — no distributional
|
|
189
|
-
* assumptions, robust to skew, easy to reason about.
|
|
190
|
-
*/
|
|
191
|
-
declare function bootstrapCi(baseline: number[], candidate: number[], options?: BootstrapOptions): BootstrapResult;
|
|
192
|
-
/**
|
|
193
|
-
* Judge-replay promotion gate.
|
|
194
|
-
*
|
|
195
|
-
* The cheap inner-loop judge that drives an evolution run is by definition
|
|
196
|
-
* fast and noisy. When you're about to promote a winning variant to the
|
|
197
|
-
* canonical default, you want a STRONGER judge (a more expensive model, a
|
|
198
|
-
* human grader, a separately-trained reward model) to confirm the win
|
|
199
|
-
* generalises beyond the inner loop.
|
|
200
|
-
*
|
|
201
|
-
* This helper takes raw winner + baseline outputs, scores both through the
|
|
202
|
-
* stronger judge, and applies `bootstrapCi`. ADVANCE means the stronger
|
|
203
|
-
* judge agrees the winner is real with the configured confidence. Doesn't
|
|
204
|
-
* matter what shape your "output" is — pass a string, an object, anything
|
|
205
|
-
* the judge can read.
|
|
206
|
-
*/
|
|
207
|
-
interface JudgeReplayGateArgs<TOutput> {
|
|
208
|
-
baselineOutputs: TOutput[];
|
|
209
|
-
candidateOutputs: TOutput[];
|
|
210
|
-
/** Stronger judge — async to allow LLM calls. Return a 0..N scalar score. */
|
|
211
|
-
judge: (output: TOutput) => Promise<number> | number;
|
|
212
|
-
alpha?: number;
|
|
213
|
-
iterations?: number;
|
|
214
|
-
/** RNG seed for reproducibility. */
|
|
215
|
-
seed?: number;
|
|
216
|
-
/** Maximum concurrent judge calls. Default 4. */
|
|
217
|
-
judgeConcurrency?: number;
|
|
218
|
-
}
|
|
219
|
-
/**
|
|
220
|
-
* Confirm a candidate's win with a stronger judge: score baseline and candidate outputs independently, then bootstrap a CI to verify the lift generalises beyond the inner loop.
|
|
221
|
-
*/
|
|
222
|
-
declare function judgeReplayGate<TOutput>(args: JudgeReplayGateArgs<TOutput>): Promise<BootstrapResult & {
|
|
223
|
-
baselineSamples: number;
|
|
224
|
-
candidateSamples: number;
|
|
225
|
-
}>;
|
|
226
|
-
|
|
227
|
-
interface RenderReleaseReportOptions {
|
|
228
|
-
title?: string;
|
|
229
|
-
runs?: readonly RunRecord[];
|
|
230
|
-
comparator?: string;
|
|
231
|
-
traceAnalystFindings?: readonly string[];
|
|
232
|
-
nextActions?: readonly string[];
|
|
233
|
-
}
|
|
234
|
-
declare function renderReleaseReport(scorecard: ReleaseConfidenceScorecard, options?: RenderReleaseReportOptions): string;
|
|
235
|
-
|
|
236
|
-
export { type ActionableSideInfo as A, type BootstrapOptions as B, type JudgeReplayGateArgs as J, type ReleaseConfidenceAxis as R, type Verdict as V, type BootstrapResult as a, type ReleaseConfidenceAxisName as b, type ReleaseConfidenceInput as c, type ReleaseConfidenceIssue as d, type ReleaseConfidenceMetrics as e, type ReleaseConfidenceScorecard as f, type ReleaseConfidenceStatus as g, type ReleaseConfidenceThresholds as h, type ReleaseTraceEvidence as i, type RenderReleaseReportOptions as j, assertReleaseConfidence as k, bootstrapCi as l, evaluateReleaseConfidence as m, judgeReplayGate as n, type AsiSeverity as o, renderReleaseReport as r };
|