@tangle-network/agent-eval 0.117.1 → 0.118.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +23 -0
- package/dist/analyst/index.d.ts +2772 -21
- package/dist/analyst/index.js +7 -6
- package/dist/analyst/index.js.map +1 -1
- package/dist/belief-state/index.d.ts +706 -10
- package/dist/belief-state/index.js +2 -1
- package/dist/belief-state/index.js.map +1 -1
- package/dist/benchmarks/index.d.ts +958 -14
- package/dist/benchmarks/index.js +12 -10
- package/dist/builder-eval/index.d.ts +449 -4
- package/dist/builder-eval/index.js +4 -3
- package/dist/builder-eval/index.js.map +1 -1
- package/dist/campaign/index.d.ts +4275 -73
- package/dist/campaign/index.js +12 -10
- package/dist/{chunk-VF3XSYTI.js → chunk-33JA4TFA.js} +6 -6
- package/dist/{chunk-4JLWXDYA.js → chunk-3EHHMC6E.js} +2 -2
- package/dist/{chunk-CCZIVI3F.js → chunk-BFW56GTT.js} +2 -2
- package/dist/{chunk-YZPO4UHR.js → chunk-DP3BBMYJ.js} +100 -143
- package/dist/chunk-DP3BBMYJ.js.map +1 -0
- package/dist/{chunk-E4BUPP7Z.js → chunk-HKUCJ437.js} +38 -66
- package/dist/chunk-HKUCJ437.js.map +1 -0
- package/dist/{chunk-HZHNRYHK.js → chunk-K6N6XJJX.js} +2 -2
- package/dist/chunk-KSDQVPLR.js +286 -0
- package/dist/chunk-KSDQVPLR.js.map +1 -0
- package/dist/chunk-MA6HLL3S.js +65 -0
- package/dist/chunk-MA6HLL3S.js.map +1 -0
- package/dist/{chunk-MGEHEHSN.js → chunk-OIIMMLRB.js} +11 -11
- package/dist/{chunk-DXZRATT5.js → chunk-OYZAPX5G.js} +3 -3
- package/dist/chunk-PXE2VKMX.js +140 -0
- package/dist/chunk-PXE2VKMX.js.map +1 -0
- package/dist/{chunk-JSJZ4PJ6.js → chunk-Q442S5AS.js} +17 -17
- package/dist/{chunk-ODVOOEWQ.js → chunk-QBRSJK47.js} +2 -2
- package/dist/{chunk-S2F4J57L.js → chunk-QKEGNI5B.js} +77 -32
- package/dist/chunk-QKEGNI5B.js.map +1 -0
- package/dist/{chunk-5UF54T55.js → chunk-S3UZOQ5Y.js} +34 -7
- package/dist/chunk-S3UZOQ5Y.js.map +1 -0
- package/dist/{chunk-FQNLDL4D.js → chunk-SVH2ANFD.js} +136 -4
- package/dist/chunk-SVH2ANFD.js.map +1 -0
- package/dist/{chunk-GQCZRZ7L.js → chunk-U5CHZ5M3.js} +9 -9
- package/dist/{chunk-TVVP3ZZQ.js → chunk-VQMK5FMP.js} +2 -1
- package/dist/{chunk-TVVP3ZZQ.js.map → chunk-VQMK5FMP.js.map} +1 -1
- package/dist/chunk-WDHBCA3M.js +31 -0
- package/dist/chunk-WDHBCA3M.js.map +1 -0
- package/dist/{chunk-HQPHZGL6.js → chunk-YLMUS4MM.js} +9 -9
- package/dist/{chunk-LQUTGLOZ.js → chunk-ZET2UAYW.js} +15 -65
- package/dist/chunk-ZET2UAYW.js.map +1 -0
- package/dist/contract/index.d.ts +4012 -38
- package/dist/contract/index.js +56 -29
- package/dist/contract/index.js.map +1 -1
- package/dist/control.d.ts +1013 -9
- package/dist/control.js +4 -3
- package/dist/fuzz.d.ts +194 -4
- package/dist/fuzz.js +3 -3
- package/dist/hosted/index.d.ts +498 -17
- package/dist/index.d.ts +10982 -1286
- package/dist/index.js +97 -80
- package/dist/index.js.map +1 -1
- package/dist/matrix/index.d.ts +139 -4
- package/dist/meta-eval/index.d.ts +862 -15
- package/dist/meta-eval/index.js +2 -1
- package/dist/meta-eval/index.js.map +1 -1
- package/dist/multishot/index.d.ts +214 -14
- package/dist/openapi.json +1 -1
- package/dist/pipelines/index.d.ts +392 -7
- package/dist/pipelines/index.js +5 -3
- package/dist/pipelines/index.js.map +1 -1
- package/dist/reporting.d.ts +1277 -17
- package/dist/rl.d.ts +2359 -28
- package/dist/rl.js +6 -5
- package/dist/rl.js.map +1 -1
- package/dist/storyboard/index.d.ts +86 -1
- package/dist/trace-attributes.d.ts +16 -0
- package/dist/trace-attributes.js +32 -0
- package/dist/trace-attributes.js.map +1 -0
- package/dist/traces.d.ts +1970 -697
- package/dist/traces.js +49 -32
- package/dist/wire/index.d.ts +655 -9
- package/docs/insight-report.md +44 -0
- package/package.json +9 -3
- package/dist/adversarial-B7loGVVX.d.ts +0 -19
- package/dist/analyst-C8HHvfJp.d.ts +0 -88
- package/dist/analyze-runs--2x39HZ7.d.ts +0 -81
- package/dist/baseline-DKq3gJpP.d.ts +0 -141
- package/dist/calibration-C8MTS7cw.d.ts +0 -101
- package/dist/chunk-5UF54T55.js.map +0 -1
- package/dist/chunk-E4BUPP7Z.js.map +0 -1
- package/dist/chunk-FQNLDL4D.js.map +0 -1
- package/dist/chunk-LQUTGLOZ.js.map +0 -1
- package/dist/chunk-S2F4J57L.js.map +0 -1
- package/dist/chunk-YZPO4UHR.js.map +0 -1
- package/dist/code-agent-session-CjZsVd19.d.ts +0 -87
- package/dist/control-6vuGfmDH.d.ts +0 -258
- package/dist/cost-ledger-DWy3XdJc.d.ts +0 -183
- package/dist/dataset-NENEzRgk.d.ts +0 -115
- package/dist/default-registry-DaK8b3fv.d.ts +0 -155
- package/dist/emitter-CjD7vUwv.d.ts +0 -122
- package/dist/errors-oeQrLqXC.d.ts +0 -74
- package/dist/failure-cluster-DOAcSJ87.d.ts +0 -76
- package/dist/feedback-trajectory-BUnM58xL.d.ts +0 -348
- package/dist/gepa-eESocoDi.d.ts +0 -642
- package/dist/index-PdX4VnPA.d.ts +0 -423
- package/dist/insight-report-DY4nDW9Q.d.ts +0 -310
- package/dist/integrity-DqlBiLyK.d.ts +0 -81
- package/dist/judge-calibration-7C-IDmKr.d.ts +0 -145
- package/dist/kind-factory-ClZmO25A.d.ts +0 -171
- package/dist/llm-client-qoDd18Qz.d.ts +0 -289
- package/dist/multi-layer-verifier-BsqKuLyN.d.ts +0 -150
- package/dist/off-policy-DiwuKKg7.d.ts +0 -132
- package/dist/outcome-store-rnXLEqSn.d.ts +0 -63
- package/dist/policy-edit-wG9uFEFm.d.ts +0 -455
- package/dist/pre-registration-BWQhJ3vz.d.ts +0 -761
- package/dist/provenance-DpjwyseI.d.ts +0 -541
- package/dist/query-CF7PG61p.d.ts +0 -35
- package/dist/raw-provider-sink-C46HDghv.d.ts +0 -132
- package/dist/release-report-C8G2i5Xi.d.ts +0 -236
- package/dist/researcher-C8XyxQsu.d.ts +0 -387
- package/dist/rubric-predictive-validity-p49lLVrE.d.ts +0 -105
- package/dist/run-record-BDH49H2E.d.ts +0 -360
- package/dist/runtime-trajectory-DGBIUt4B.d.ts +0 -49
- package/dist/schema-B3Q3l9Z_.d.ts +0 -201
- package/dist/semantic-concept-judge-CXnPEJbf.d.ts +0 -723
- package/dist/sequential-5iSVfzl2.d.ts +0 -139
- package/dist/series-convergence-D5OWMBg6.d.ts +0 -33
- package/dist/statistics-KUnG73jH.d.ts +0 -494
- package/dist/storage-DrX3v_5B.d.ts +0 -50
- package/dist/store-C1YxJDEK.d.ts +0 -248
- package/dist/store-DGqD0Pyo.d.ts +0 -116
- package/dist/summary-report-C5bKFfm-.d.ts +0 -445
- package/dist/test-graded-scenario-B0ybnPY7.d.ts +0 -166
- package/dist/types-BSw1rOUB.d.ts +0 -634
- package/dist/types-BUxNaJ8c.d.ts +0 -108
- package/dist/types-BkfcQnxV.d.ts +0 -313
- package/dist/verdict-C9MlYujm.d.ts +0 -35
- /package/dist/{chunk-VF3XSYTI.js.map → chunk-33JA4TFA.js.map} +0 -0
- /package/dist/{chunk-4JLWXDYA.js.map → chunk-3EHHMC6E.js.map} +0 -0
- /package/dist/{chunk-CCZIVI3F.js.map → chunk-BFW56GTT.js.map} +0 -0
- /package/dist/{chunk-HZHNRYHK.js.map → chunk-K6N6XJJX.js.map} +0 -0
- /package/dist/{chunk-MGEHEHSN.js.map → chunk-OIIMMLRB.js.map} +0 -0
- /package/dist/{chunk-DXZRATT5.js.map → chunk-OYZAPX5G.js.map} +0 -0
- /package/dist/{chunk-JSJZ4PJ6.js.map → chunk-Q442S5AS.js.map} +0 -0
- /package/dist/{chunk-ODVOOEWQ.js.map → chunk-QBRSJK47.js.map} +0 -0
- /package/dist/{chunk-GQCZRZ7L.js.map → chunk-U5CHZ5M3.js.map} +0 -0
- /package/dist/{chunk-HQPHZGL6.js.map → chunk-YLMUS4MM.js.map} +0 -0
package/dist/contract/index.d.ts
CHANGED
|
@@ -1,39 +1,3584 @@
|
|
|
1
|
-
import {
|
|
2
|
-
|
|
3
|
-
|
|
4
|
-
|
|
5
|
-
|
|
6
|
-
|
|
7
|
-
|
|
8
|
-
|
|
9
|
-
|
|
10
|
-
|
|
11
|
-
|
|
12
|
-
|
|
13
|
-
|
|
14
|
-
|
|
15
|
-
|
|
16
|
-
|
|
17
|
-
|
|
18
|
-
|
|
19
|
-
|
|
20
|
-
|
|
21
|
-
|
|
22
|
-
|
|
23
|
-
|
|
24
|
-
|
|
25
|
-
|
|
26
|
-
|
|
27
|
-
|
|
28
|
-
|
|
29
|
-
|
|
30
|
-
|
|
31
|
-
|
|
32
|
-
|
|
33
|
-
|
|
34
|
-
|
|
35
|
-
|
|
36
|
-
|
|
1
|
+
import { AxFunction, AxAIService } from '@ax-llm/ax';
|
|
2
|
+
import { z } from 'zod';
|
|
3
|
+
|
|
4
|
+
type AgentProfileCellSchemaVersion = 'agent-profile-cell/v1';
|
|
5
|
+
type AgentProfileJson = string | number | boolean | null | AgentProfileJson[] | {
|
|
6
|
+
[key: string]: AgentProfileJson;
|
|
7
|
+
};
|
|
8
|
+
type AgentProfileDimensionValue = string | number | boolean | null;
|
|
9
|
+
interface AgentProfileSource {
|
|
10
|
+
/** Runtime/profile contract being fingerprinted, e.g. `agent-interface-profile`. */
|
|
11
|
+
kind: string;
|
|
12
|
+
/** sha256 over the canonical source profile object. */
|
|
13
|
+
hash: string;
|
|
14
|
+
}
|
|
15
|
+
interface AgentProfileHarness {
|
|
16
|
+
id: string;
|
|
17
|
+
version?: string;
|
|
18
|
+
hash?: string;
|
|
19
|
+
}
|
|
20
|
+
interface AgentProfileCell {
|
|
21
|
+
schemaVersion: AgentProfileCellSchemaVersion;
|
|
22
|
+
cellId: string;
|
|
23
|
+
profileId: string;
|
|
24
|
+
sourceProfile: AgentProfileSource;
|
|
25
|
+
harness?: AgentProfileHarness;
|
|
26
|
+
model?: string;
|
|
27
|
+
promptHash?: string;
|
|
28
|
+
dimensions?: Record<string, AgentProfileDimensionValue>;
|
|
29
|
+
}
|
|
30
|
+
|
|
31
|
+
type FailureClass = 'success' | 'reasoning_error' | 'tool_selection_error' | 'tool_argument_error' | 'tool_recovery_failure' | 'hallucination' | 'instruction_following' | 'safety_refusal_miss' | 'policy_violation' | 'budget_exceeded' | 'format_drift' | 'permission_escalation' | 'pii_leak' | 'cost_overrun' | 'timeout' | 'sandbox_failure' | 'missing_user_data' | 'missing_domain_data' | 'missing_codebase_context' | 'missing_runtime_context' | 'missing_credentials' | 'missing_integration_connection' | 'missing_integration_scope' | 'integration_approval_required' | 'integration_auth_expired' | 'integration_provider_failure' | 'bad_integration_manifest' | 'unsafe_integration_write_denied' | 'stale_external_data' | 'bad_retrieval' | 'insufficient_evidence' | 'contradictory_evidence' | 'ambiguous_user_intent' | 'knowledge_readiness_blocked' | 'unknown';
|
|
32
|
+
|
|
33
|
+
/**
|
|
34
|
+
* Paper-grade RunRecord schema + runtime validator.
|
|
35
|
+
*
|
|
36
|
+
* Every run that participates in a promotion gate, paper table, or
|
|
37
|
+
* researcher loop SHOULD be recorded as a `RunRecord`. The mandatory
|
|
38
|
+
* fields are exactly those the paper "Two Loops, Three Roles" requires
|
|
39
|
+
* for reproducibility: who/what/when/cost/seed/hash, plus the search vs
|
|
40
|
+
* holdout split tag and either a `searchScore` or a `holdoutScore`.
|
|
41
|
+
*
|
|
42
|
+
* This is intentionally NOT a replacement for the rich `Run` /
|
|
43
|
+
* `ProposeReviewReport` / `ScenarioResult` types already in the
|
|
44
|
+
* package. Those are runtime structures with full provenance. A
|
|
45
|
+
* `RunRecord` is the analysis-time projection — the JSON-friendly
|
|
46
|
+
* row you'd put in a parquet file or paste into a notebook.
|
|
47
|
+
*
|
|
48
|
+
* Validate at the boundary:
|
|
49
|
+
*
|
|
50
|
+
* const rec = validateRunRecord(rawJson) // throws on missing
|
|
51
|
+
* const ok = isRunRecord(rawJson) // boolean check
|
|
52
|
+
* const rec = parseRunRecordSafe(rawJson) // { ok, value | error }
|
|
53
|
+
*
|
|
54
|
+
* The validator runs in pure TS — zod is intentionally NOT a
|
|
55
|
+
* dependency. Round-trip tested in `tests/run-record.test.ts`.
|
|
56
|
+
*/
|
|
57
|
+
|
|
58
|
+
/** Search/dev/holdout split tag. 'search' is the paper-grade alias for the
|
|
59
|
+
* combined train+test pool that the optimizer is allowed to read. */
|
|
60
|
+
type RunSplitTag = 'search' | 'dev' | 'holdout';
|
|
61
|
+
interface RunTokenUsage {
|
|
62
|
+
input: number;
|
|
63
|
+
/** All generated tokens charged as output, including reasoning tokens. */
|
|
64
|
+
output: number;
|
|
65
|
+
/** Reasoning-token subset of `output`, when the provider reports it. */
|
|
66
|
+
reasoning?: number;
|
|
67
|
+
/** Prompt tokens served from a provider cache. */
|
|
68
|
+
cached?: number;
|
|
69
|
+
/** Prompt tokens written into a provider cache. */
|
|
70
|
+
cacheWrite?: number;
|
|
71
|
+
}
|
|
72
|
+
/**
|
|
73
|
+
* How a run's USD amount was obtained.
|
|
74
|
+
*
|
|
75
|
+
* `costUsd` remains mandatory for wire compatibility. New producers should
|
|
76
|
+
* always populate this discriminated union so a missing bill is never
|
|
77
|
+
* mistaken for an observed zero-dollar run. For `uncaptured`, `costUsd` uses
|
|
78
|
+
* the legacy `0` sentinel while this field carries the truthful null.
|
|
79
|
+
*/
|
|
80
|
+
type RunCostProvenance = {
|
|
81
|
+
kind: 'observed';
|
|
82
|
+
usd: number;
|
|
83
|
+
} | {
|
|
84
|
+
kind: 'estimated';
|
|
85
|
+
usd: number;
|
|
86
|
+
} | {
|
|
87
|
+
kind: 'uncaptured';
|
|
88
|
+
usd: null;
|
|
89
|
+
};
|
|
90
|
+
interface RunJudgeMetadata {
|
|
91
|
+
model: string;
|
|
92
|
+
promptVersion: string;
|
|
93
|
+
/** [0,1] confidence the judge declared. Constant judge confidence
|
|
94
|
+
* across many runs is a fallback signal (see `canary.ts`). */
|
|
95
|
+
confidence: number;
|
|
96
|
+
/** True if the judge degraded to a fallback path (rules-only,
|
|
97
|
+
* prior-call cache, etc.). The canary uses this to alert. */
|
|
98
|
+
fallback: boolean;
|
|
99
|
+
}
|
|
100
|
+
/**
|
|
101
|
+
* Per-judge / per-dimension breakdown for runs scored by an ensemble of
|
|
102
|
+
* judges over a multi-dimensional rubric.
|
|
103
|
+
*
|
|
104
|
+
* The collapsed `outcome.searchScore` / `holdoutScore` carries the
|
|
105
|
+
* composite the gate uses. The full breakdown belongs here so consumers
|
|
106
|
+
* can answer "which judge disagreed?", "which dimension dragged the
|
|
107
|
+
* composite down?", and "did half the panel fail?" without re-running.
|
|
108
|
+
*
|
|
109
|
+
* `perJudge[judgeId][dim]` is the canonical source; `perDimMean` and
|
|
110
|
+
* `composite` are convenience projections — derivable but precomputed so
|
|
111
|
+
* downstream IRR primitives (`interRaterReliability`,
|
|
112
|
+
* `corpusInterRaterAgreement`) and reporters don't pay the same
|
|
113
|
+
* aggregation twice.
|
|
114
|
+
*
|
|
115
|
+
* Fail-loud discipline: judges that errored out land in `failedJudges`
|
|
116
|
+
* by id. A missing key in `perJudge` is ambiguous (silent zero vs not
|
|
117
|
+
* run); the explicit list makes a partial-failure recorded as such.
|
|
118
|
+
*/
|
|
119
|
+
interface JudgeScoresRecord {
|
|
120
|
+
/** Per-judge per-dimension scores. `{ "kimi-k2.6": { helpfulness: 0.8, clarity: 0.7 }, ... }`. */
|
|
121
|
+
perJudge: Record<string, Record<string, number>>;
|
|
122
|
+
/** Per-dim mean across judges. Convenience — derivable from `perJudge`. */
|
|
123
|
+
perDimMean: Record<string, number>;
|
|
124
|
+
/** Composite mean across all dims and judges. Mirrors the score
|
|
125
|
+
* the gate sees on `outcome.searchScore` / `holdoutScore`. */
|
|
126
|
+
composite: number;
|
|
127
|
+
/** Judges that errored or returned an unparseable verdict. Recorded
|
|
128
|
+
* by id (e.g. `['glm-5.1']`) so a partial-failure case is explicit,
|
|
129
|
+
* not inferred from missing keys in `perJudge`. */
|
|
130
|
+
failedJudges?: string[];
|
|
131
|
+
/** Free-form notes the judges emitted (joined across judges or
|
|
132
|
+
* first-judge only — consumer's choice). */
|
|
133
|
+
notes?: string;
|
|
134
|
+
}
|
|
135
|
+
interface RunOutcome {
|
|
136
|
+
/** Score on the search/optimization split. Optional because a
|
|
137
|
+
* holdout-only evaluation only fills `holdoutScore`. */
|
|
138
|
+
searchScore?: number;
|
|
139
|
+
/** Score on the held-out split. Optional because a search-only run
|
|
140
|
+
* only fills `searchScore`. At least one must be present. */
|
|
141
|
+
holdoutScore?: number;
|
|
142
|
+
/** Bag of any other metric the run produced — judge dimensions,
|
|
143
|
+
* pass/fail counters, latency stats, etc. Numeric only — keeps
|
|
144
|
+
* reporters honest. */
|
|
145
|
+
raw: Record<string, number>;
|
|
146
|
+
/** Per-judge / per-dim breakdown. Consumers writing ensemble
|
|
147
|
+
* judgements populate this; substrate primitives like
|
|
148
|
+
* `interRaterReliability` and `corpusInterRaterAgreement` accept
|
|
149
|
+
* these records as input. Optional — single-judge or scalar-only
|
|
150
|
+
* runs leave it unset. */
|
|
151
|
+
judgeScores?: JudgeScoresRecord;
|
|
152
|
+
/** Authenticity / realness verdict — did the run build the REAL thing on the
|
|
153
|
+
* intended infra, or fake it (see `./authenticity`)? Optional: only domains
|
|
154
|
+
* with an authenticity config populate it. Carried in the corpus so the
|
|
155
|
+
* flywheel / off-policy learning can optimize for real completion, not gamed
|
|
156
|
+
* pass-rate. `score` is 0-1; `gated` is the anti-Goodhart flag — a gated run
|
|
157
|
+
* must not count as a real success regardless of `score`. */
|
|
158
|
+
realness?: {
|
|
159
|
+
score: number;
|
|
160
|
+
gated: boolean;
|
|
161
|
+
reason?: string;
|
|
162
|
+
};
|
|
163
|
+
}
|
|
164
|
+
/**
|
|
165
|
+
* Mandatory paper-grade fields for a single evaluation run. Optional
|
|
166
|
+
* fields are extension points; mandatory fields throw if missing.
|
|
167
|
+
*
|
|
168
|
+
* Hash discipline:
|
|
169
|
+
* - `promptHash` is the sha256 of the EFFECTIVE prompt sent to the
|
|
170
|
+
* model (after any steering bundle merge).
|
|
171
|
+
* - `configHash` is the sha256 of the effective run config (model,
|
|
172
|
+
* temperature, tools, judges, splits). The pair (promptHash,
|
|
173
|
+
* configHash) uniquely identifies an experiment cell.
|
|
174
|
+
*
|
|
175
|
+
* Model snapshot discipline:
|
|
176
|
+
* - `model` MUST encode a snapshot version. Bare aliases like
|
|
177
|
+
* `claude-sonnet-4` or `gpt-4o` are banned — they remap silently.
|
|
178
|
+
* Use `claude-sonnet-4-6@2025-04-15` or `gpt-4o-2024-11-20`.
|
|
179
|
+
*/
|
|
180
|
+
interface RunRecord {
|
|
181
|
+
/** UUID for the run. */
|
|
182
|
+
runId: string;
|
|
183
|
+
/** Logical experiment grouping (a treatment vs a baseline within
|
|
184
|
+
* the same sweep should share `experimentId`). */
|
|
185
|
+
experimentId: string;
|
|
186
|
+
/** Stable identifier for the candidate (variant) being run. The
|
|
187
|
+
* promotion gate compares two `candidateId`s on matched items. */
|
|
188
|
+
candidateId: string;
|
|
189
|
+
/** RNG seed for the run. Always recorded — silent re-seeding is
|
|
190
|
+
* the most common cause of non-reproducible numbers. */
|
|
191
|
+
seed: number;
|
|
192
|
+
/** Model identifier WITH snapshot version. */
|
|
193
|
+
model: string;
|
|
194
|
+
/** sha256 of the effective prompt (post-steering). */
|
|
195
|
+
promptHash: string;
|
|
196
|
+
/** sha256 of the effective config. */
|
|
197
|
+
configHash: string;
|
|
198
|
+
/** Git SHA the harness was run from. */
|
|
199
|
+
commitSha: string;
|
|
200
|
+
/** End-to-end wall-clock duration in milliseconds. */
|
|
201
|
+
wallMs: number;
|
|
202
|
+
/** Time spent queued before execution started, if known. */
|
|
203
|
+
queueMs?: number;
|
|
204
|
+
/** Total USD cost. Mandatory — runs without a cost number are
|
|
205
|
+
* unbounded by definition and must not be admitted into the gate.
|
|
206
|
+
* `0` is retained as the compatibility sentinel for an uncaptured amount;
|
|
207
|
+
* inspect `costProvenance` before treating it as observed. */
|
|
208
|
+
costUsd: number;
|
|
209
|
+
/** Observed, model-priced estimate, or genuinely uncaptured USD amount.
|
|
210
|
+
* Optional only so existing serialized RunRecords remain valid. */
|
|
211
|
+
costProvenance?: RunCostProvenance;
|
|
212
|
+
/** Token usage breakdown. */
|
|
213
|
+
tokenUsage: RunTokenUsage;
|
|
214
|
+
/** Judge-side metadata, if a judge was used. */
|
|
215
|
+
judgeMetadata?: RunJudgeMetadata;
|
|
216
|
+
/** Per-split scores + raw bag. */
|
|
217
|
+
outcome: RunOutcome;
|
|
218
|
+
/** Canonical, cross-agent failure class drawn from the shared
|
|
219
|
+
* `FAILURE_CLASSES` taxonomy. This is the aggregation key that makes
|
|
220
|
+
* "which failure dominates across the whole fleet" answerable in ONE
|
|
221
|
+
* vocabulary — every agent classifies against the same enum. Producers
|
|
222
|
+
* set it via the substrate classifier; leave unset only when the failure
|
|
223
|
+
* genuinely can't be classified. */
|
|
224
|
+
failureClass?: FailureClass;
|
|
225
|
+
/** Free-form domain-specific failure detail, scoped UNDER `failureClass`
|
|
226
|
+
* (e.g. failureClass='tool_recovery_failure', failureMode='forge_build_unsatisfied').
|
|
227
|
+
* The within-agent drill-down; `failureClass` is the cross-agent key. */
|
|
228
|
+
failureMode?: string;
|
|
229
|
+
/** Which split this run was drawn from. */
|
|
230
|
+
splitTag: RunSplitTag;
|
|
231
|
+
/**
|
|
232
|
+
* Stable scenario identifier the run was scored against. Optional for
|
|
233
|
+
* backwards compatibility, but **strongly recommended**: every primitive
|
|
234
|
+
* that pairs runs by scenario (preferences, paired stats, BT tournament)
|
|
235
|
+
* keys on this. The campaign artifact populates it canonically; legacy
|
|
236
|
+
* runs without it fall back to inference from `outcome.raw.scenario_id`
|
|
237
|
+
* or `experimentId`.
|
|
238
|
+
*/
|
|
239
|
+
scenarioId?: string;
|
|
240
|
+
/**
|
|
241
|
+
* Canonical identity for the agent profile cell that produced this row:
|
|
242
|
+
* profile artifact hash plus optional harness/model/prompt/reporting
|
|
243
|
+
* dimensions. Use `agentProfile.cellId` to group persona sweeps and
|
|
244
|
+
* longitudinal reports by the complete source profile, not by a loose
|
|
245
|
+
* candidate label or opaque config hash.
|
|
246
|
+
*/
|
|
247
|
+
agentProfile?: AgentProfileCell;
|
|
248
|
+
}
|
|
249
|
+
|
|
250
|
+
/**
|
|
251
|
+
* Shared types for the trace-analyst module.
|
|
252
|
+
*
|
|
253
|
+
* Wire format. The store interface speaks `OtlpSpanLike` rows — one JSONL
|
|
254
|
+
* line per span, OTLP-shaped. We do NOT depend on a specific tracing
|
|
255
|
+
* vendor at the type level. Adapter
|
|
256
|
+
* layers map upstream shapes onto this interface.
|
|
257
|
+
*
|
|
258
|
+
* Design constraint. Every read operation that can return arbitrary
|
|
259
|
+
* payload must carry a byte budget so the agent's tool result stays
|
|
260
|
+
* bounded regardless of input trace size. Oversized responses
|
|
261
|
+
* substitute a deterministic summary instead of bytes — see
|
|
262
|
+
* `ViewTraceOversized`.
|
|
263
|
+
*/
|
|
264
|
+
/** OTLP span kind (subset we actually use). */
|
|
265
|
+
type TraceAnalystSpanKind = 'AGENT' | 'LLM' | 'TOOL' | 'CHAIN' | 'GUARDRAIL' | 'SPAN' | 'UNKNOWN';
|
|
266
|
+
type TraceAnalystSpanStatus = 'OK' | 'ERROR' | 'UNSET';
|
|
267
|
+
/** Subset of OTLP span fields the analyst exposes to the agent. The
|
|
268
|
+
* store's job is to project upstream's full span shape down to this
|
|
269
|
+
* view — the analyst never sees vendor extensions directly. */
|
|
270
|
+
interface TraceAnalystSpan {
|
|
271
|
+
trace_id: string;
|
|
272
|
+
span_id: string;
|
|
273
|
+
parent_span_id: string | null;
|
|
274
|
+
name: string;
|
|
275
|
+
kind: TraceAnalystSpanKind;
|
|
276
|
+
start_time: string;
|
|
277
|
+
end_time: string;
|
|
278
|
+
duration_ms: number;
|
|
279
|
+
status: TraceAnalystSpanStatus;
|
|
280
|
+
status_message?: string;
|
|
281
|
+
service_name: string | null;
|
|
282
|
+
agent_name: string | null;
|
|
283
|
+
model_name: string | null;
|
|
284
|
+
tool_name: string | null;
|
|
285
|
+
/** Raw JSON-serialisable attribute map. May contain large strings;
|
|
286
|
+
* callers must respect the per-attribute byte cap. */
|
|
287
|
+
attributes: Record<string, unknown>;
|
|
288
|
+
}
|
|
289
|
+
interface TraceAnalystTraceSummary {
|
|
290
|
+
trace_id: string;
|
|
291
|
+
service_name: string | null;
|
|
292
|
+
agent_name: string | null;
|
|
293
|
+
span_count: number;
|
|
294
|
+
has_errors: boolean;
|
|
295
|
+
start_time: string;
|
|
296
|
+
end_time: string;
|
|
297
|
+
duration_ms: number;
|
|
298
|
+
raw_jsonl_bytes: number;
|
|
299
|
+
models: string[];
|
|
300
|
+
tools: string[];
|
|
301
|
+
}
|
|
302
|
+
interface TraceAnalystFilters {
|
|
303
|
+
/** Restrict to traces that contain at least one error span. */
|
|
304
|
+
has_errors?: boolean;
|
|
305
|
+
/** Match if any span's `service.name` is in this list. */
|
|
306
|
+
service_names?: string[];
|
|
307
|
+
/** Match if any span's `agent.name` is in this list. */
|
|
308
|
+
agent_names?: string[];
|
|
309
|
+
/** Match if any LLM span's `llm.model_name` is in this list. */
|
|
310
|
+
model_names?: string[];
|
|
311
|
+
/** Match if any tool span's `tool.name` is in this list. */
|
|
312
|
+
tool_names?: string[];
|
|
313
|
+
/** ISO-8601 lower bound on the trace's earliest start time. */
|
|
314
|
+
start_time_after?: string;
|
|
315
|
+
/** ISO-8601 upper bound on the trace's earliest start time. */
|
|
316
|
+
start_time_before?: string;
|
|
317
|
+
/** Single regex applied to raw JSONL bytes for the trace. Opt-in;
|
|
318
|
+
* expensive on large datasets. Use the indexed filters above first. */
|
|
319
|
+
regex_pattern?: string;
|
|
320
|
+
}
|
|
321
|
+
/** One distinct error signature across the dataset — the deterministic unit of
|
|
322
|
+
* failure coverage. Signatures normalize volatile tokens (digits, hex/uuids,
|
|
323
|
+
* paths, durations) out of the span `status_message` so semantically identical
|
|
324
|
+
* failures collapse into one cluster. An analyst that accounts for every
|
|
325
|
+
* cluster has, by construction, covered every distinct failure mode. */
|
|
326
|
+
interface ErrorCluster {
|
|
327
|
+
/** Normalized status_message — the cluster key. */
|
|
328
|
+
signature: string;
|
|
329
|
+
/** A verbatim, un-normalized exemplar message (for exact-string citation). */
|
|
330
|
+
status_message_sample: string;
|
|
331
|
+
/** The span name that most often carries this signature, if any. */
|
|
332
|
+
span_name: string | null;
|
|
333
|
+
/** The tool that most often carries this signature, if any. */
|
|
334
|
+
tool_name: string | null;
|
|
335
|
+
trace_count: number;
|
|
336
|
+
span_count: number;
|
|
337
|
+
/** trace_count / total error traces in the matched set (0..1). */
|
|
338
|
+
prevalence: number;
|
|
339
|
+
/** Real trace ids carrying this signature (capped), passable to view/search. */
|
|
340
|
+
exemplar_trace_ids: string[];
|
|
341
|
+
/** Real span ids carrying this signature (capped). */
|
|
342
|
+
exemplar_span_ids: string[];
|
|
343
|
+
}
|
|
344
|
+
interface DatasetOverview {
|
|
345
|
+
total_traces: number;
|
|
346
|
+
raw_jsonl_bytes: number;
|
|
347
|
+
services: string[];
|
|
348
|
+
agents: string[];
|
|
349
|
+
models: string[];
|
|
350
|
+
tool_names: string[];
|
|
351
|
+
/** Up to 20 real trace ids the agent may pass to view/search tools. */
|
|
352
|
+
sample_trace_ids: string[];
|
|
353
|
+
errors: {
|
|
354
|
+
trace_count: number;
|
|
355
|
+
span_count: number;
|
|
356
|
+
};
|
|
357
|
+
/** The COMPLETE deterministic error-signature population, sorted by
|
|
358
|
+
* trace_count desc. This is the failure-coverage checklist: an analysis is
|
|
359
|
+
* complete only when every cluster here is accounted for. Empty when the
|
|
360
|
+
* matched set has no error spans. */
|
|
361
|
+
error_clusters: ErrorCluster[];
|
|
362
|
+
time_range: {
|
|
363
|
+
earliest: string;
|
|
364
|
+
latest: string;
|
|
365
|
+
} | null;
|
|
366
|
+
}
|
|
367
|
+
interface QueryTracesPage {
|
|
368
|
+
traces: TraceAnalystTraceSummary[];
|
|
369
|
+
total: number;
|
|
370
|
+
has_more: boolean;
|
|
371
|
+
}
|
|
372
|
+
/** Full-trace view. When the response would exceed the per-call byte
|
|
373
|
+
* budget, `oversized` is populated INSTEAD of `spans` so the agent
|
|
374
|
+
* knows to switch to `searchTrace` / `viewSpans`. */
|
|
375
|
+
interface ViewTraceResult {
|
|
376
|
+
trace_id: string;
|
|
377
|
+
spans?: TraceAnalystSpan[];
|
|
378
|
+
oversized?: ViewTraceOversized;
|
|
379
|
+
}
|
|
380
|
+
interface ViewTraceOversized {
|
|
381
|
+
span_count: number;
|
|
382
|
+
/** Names with their counts, sorted desc. Capped at 20 entries. */
|
|
383
|
+
top_span_names: Array<[string, number]>;
|
|
384
|
+
/** Largest single span body (bytes after attribute-cap projection). */
|
|
385
|
+
span_response_bytes_max: number;
|
|
386
|
+
error_span_count: number;
|
|
387
|
+
}
|
|
388
|
+
interface ViewSpansResult {
|
|
389
|
+
trace_id: string;
|
|
390
|
+
spans: TraceAnalystSpan[];
|
|
391
|
+
/** Number of requested span ids that were not found in the trace. */
|
|
392
|
+
missing_span_ids: string[];
|
|
393
|
+
/** Number of attribute fields truncated to fit the per-attribute cap. */
|
|
394
|
+
truncated_attribute_count: number;
|
|
395
|
+
}
|
|
396
|
+
interface SpanMatchRecord {
|
|
397
|
+
trace_id: string;
|
|
398
|
+
span_id: string;
|
|
399
|
+
span_name: string;
|
|
400
|
+
span_kind: TraceAnalystSpanKind;
|
|
401
|
+
/** JSON pointer-style path to the matched value, e.g.
|
|
402
|
+
* `attributes."llm.input_messages"[2].content`. */
|
|
403
|
+
attribute_path: string;
|
|
404
|
+
matched_text: string;
|
|
405
|
+
context_before: string;
|
|
406
|
+
context_after: string;
|
|
407
|
+
match_offset: number;
|
|
408
|
+
}
|
|
409
|
+
interface SearchTraceResult {
|
|
410
|
+
trace_id: string;
|
|
411
|
+
hits: SpanMatchRecord[];
|
|
412
|
+
total_matches: number;
|
|
413
|
+
has_more: boolean;
|
|
414
|
+
}
|
|
415
|
+
interface SearchSpanResult {
|
|
416
|
+
trace_id: string;
|
|
417
|
+
span_id: string;
|
|
418
|
+
hits: SpanMatchRecord[];
|
|
419
|
+
total_matches: number;
|
|
420
|
+
has_more: boolean;
|
|
421
|
+
}
|
|
422
|
+
|
|
423
|
+
/**
|
|
424
|
+
* `TraceAnalysisStore` — read-side interface the trace-analyst calls
|
|
425
|
+
* through. Six operations, all bounded:
|
|
426
|
+
*
|
|
427
|
+
* - `getOverview(filters?)` — dataset rollup + sample trace ids.
|
|
428
|
+
* - `queryTraces(filters?, limit, offset)` — paginated summaries.
|
|
429
|
+
* - `countTraces(filters?)` — cheap count without materialisation.
|
|
430
|
+
* - `viewTrace(trace_id, perAttrCap)` — full span list, oversized → summary.
|
|
431
|
+
* - `viewSpans(trace_id, span_ids, perAttrCap)` — surgical span fetch.
|
|
432
|
+
* - `searchTrace(trace_id, regex, max_matches)` — bounded regex hits.
|
|
433
|
+
* - `searchSpan(trace_id, span_id, regex, max_matches)` — single-span search.
|
|
434
|
+
*
|
|
435
|
+
* Multiple implementations ship in the core (`OtlpFileTraceStore`).
|
|
436
|
+
* Downstream callers can supply their own — e.g. a DuckDB-backed
|
|
437
|
+
* adapter or an in-memory adapter for tests — by implementing this
|
|
438
|
+
* interface.
|
|
439
|
+
*
|
|
440
|
+
* Filters compose with AND semantics. Empty/undefined fields impose
|
|
441
|
+
* no constraint. `regex_pattern` is the only opt-in raw-bytes scan —
|
|
442
|
+
* implementations may skip it via `count`/`overview` when not set.
|
|
443
|
+
*/
|
|
444
|
+
|
|
445
|
+
interface TraceAnalysisStore {
|
|
446
|
+
getOverview(filters?: TraceAnalystFilters): Promise<DatasetOverview>;
|
|
447
|
+
queryTraces(opts: {
|
|
448
|
+
filters?: TraceAnalystFilters;
|
|
449
|
+
limit: number;
|
|
450
|
+
offset?: number;
|
|
451
|
+
}): Promise<QueryTracesPage>;
|
|
452
|
+
countTraces(filters?: TraceAnalystFilters): Promise<number>;
|
|
453
|
+
viewTrace(opts: {
|
|
454
|
+
trace_id: string;
|
|
455
|
+
/** Override per-attribute byte cap. Defaults to discovery budget. */
|
|
456
|
+
per_attribute_byte_cap?: number;
|
|
457
|
+
}): Promise<ViewTraceResult>;
|
|
458
|
+
viewSpans(opts: {
|
|
459
|
+
trace_id: string;
|
|
460
|
+
span_ids: readonly string[];
|
|
461
|
+
/** Override per-attribute byte cap. Defaults to surgical budget. */
|
|
462
|
+
per_attribute_byte_cap?: number;
|
|
463
|
+
}): Promise<ViewSpansResult>;
|
|
464
|
+
searchTrace(opts: {
|
|
465
|
+
trace_id: string;
|
|
466
|
+
regex_pattern: string;
|
|
467
|
+
/** Hard cap on matches returned. Default 50. */
|
|
468
|
+
max_matches?: number;
|
|
469
|
+
}): Promise<SearchTraceResult>;
|
|
470
|
+
searchSpan(opts: {
|
|
471
|
+
trace_id: string;
|
|
472
|
+
span_id: string;
|
|
473
|
+
regex_pattern: string;
|
|
474
|
+
max_matches?: number;
|
|
475
|
+
}): Promise<SearchSpanResult>;
|
|
476
|
+
}
|
|
477
|
+
|
|
478
|
+
type CostChannel = 'agent' | 'judge' | 'verifier' | 'analyst' | 'driver' | (string & {});
|
|
479
|
+
interface CostUsage {
|
|
480
|
+
inputTokens: number;
|
|
481
|
+
outputTokens: number;
|
|
482
|
+
cachedTokens?: number;
|
|
483
|
+
}
|
|
484
|
+
interface CostCallBase {
|
|
485
|
+
callId: string;
|
|
486
|
+
channel: CostChannel;
|
|
487
|
+
phase: string;
|
|
488
|
+
actor: string;
|
|
489
|
+
model: string;
|
|
490
|
+
maximumCostUsd?: number;
|
|
491
|
+
tags?: Record<string, string>;
|
|
492
|
+
timestamp: number;
|
|
493
|
+
}
|
|
494
|
+
interface CostReceipt extends CostCallBase, CostUsage {
|
|
495
|
+
status: 'settled';
|
|
496
|
+
costUsd: number;
|
|
497
|
+
costUnknown: boolean;
|
|
498
|
+
usageUnknown?: boolean;
|
|
499
|
+
pricing?: {
|
|
500
|
+
inputUsdPerThousand: number;
|
|
501
|
+
outputUsdPerThousand: number;
|
|
502
|
+
};
|
|
503
|
+
actualCostUsd?: number;
|
|
504
|
+
error?: string;
|
|
505
|
+
}
|
|
506
|
+
interface CostReceiptInput extends CostUsage {
|
|
507
|
+
model: string;
|
|
508
|
+
actualCostUsd?: number;
|
|
509
|
+
costUnknown?: boolean;
|
|
510
|
+
usageUnknown?: boolean;
|
|
511
|
+
}
|
|
512
|
+
type MaximumCharge = {
|
|
513
|
+
externallyEnforcedMaximumUsd: number;
|
|
514
|
+
} | ({
|
|
515
|
+
model: string;
|
|
516
|
+
} & CostUsage);
|
|
517
|
+
interface RunPaidCallInput<T> {
|
|
518
|
+
callId?: string;
|
|
519
|
+
channel: CostChannel;
|
|
520
|
+
phase: string;
|
|
521
|
+
actor: string;
|
|
522
|
+
/** Used before a provider receipt exists and on failures without one. */
|
|
523
|
+
model?: string;
|
|
524
|
+
tags?: Record<string, string>;
|
|
525
|
+
signal?: AbortSignal;
|
|
526
|
+
/** Provider-enforced dollar maximum, or maximum priced token usage. Required when capped. */
|
|
527
|
+
maximumCharge?: MaximumCharge;
|
|
528
|
+
/** `callId` can be forwarded as the provider's idempotency key. */
|
|
529
|
+
execute(signal: AbortSignal, callId: string): Promise<T>;
|
|
530
|
+
receipt(value: T): CostReceiptInput;
|
|
531
|
+
receiptFromError?(error: Error): CostReceiptInput | undefined;
|
|
532
|
+
}
|
|
533
|
+
type PaidCallResult<T> = {
|
|
534
|
+
succeeded: true;
|
|
535
|
+
callId: string;
|
|
536
|
+
value: T;
|
|
537
|
+
receipt: CostReceipt;
|
|
538
|
+
} | {
|
|
539
|
+
succeeded: false;
|
|
540
|
+
callId?: string;
|
|
541
|
+
error: Error;
|
|
542
|
+
receipt?: CostReceipt;
|
|
543
|
+
};
|
|
544
|
+
interface ChannelRollup {
|
|
545
|
+
channel: CostChannel;
|
|
546
|
+
calls: number;
|
|
547
|
+
inputTokens: number;
|
|
548
|
+
outputTokens: number;
|
|
549
|
+
cachedTokens: number;
|
|
550
|
+
costUsd: number;
|
|
551
|
+
unpricedCalls: number;
|
|
552
|
+
unknownUsageCalls: number;
|
|
553
|
+
}
|
|
554
|
+
interface CostLedgerSummary {
|
|
555
|
+
totalCalls: number;
|
|
556
|
+
pendingCalls: number;
|
|
557
|
+
unresolvedCalls: number;
|
|
558
|
+
reservedCostUsd: number;
|
|
559
|
+
inputTokens: number;
|
|
560
|
+
outputTokens: number;
|
|
561
|
+
cachedTokens: number;
|
|
562
|
+
totalCostUsd: number;
|
|
563
|
+
byChannel: ChannelRollup[];
|
|
564
|
+
unpricedModels: string[];
|
|
565
|
+
fullyPriced: boolean;
|
|
566
|
+
usageComplete: boolean;
|
|
567
|
+
accountingComplete: boolean;
|
|
568
|
+
incompleteReasons: string[];
|
|
569
|
+
}
|
|
570
|
+
interface CostLedgerFilter {
|
|
571
|
+
channel?: CostChannel;
|
|
572
|
+
phase?: string;
|
|
573
|
+
tags?: Record<string, string>;
|
|
574
|
+
}
|
|
575
|
+
/** Append-only storage. `append` must atomically reject stale revisions. */
|
|
576
|
+
interface CostLedgerPersistence {
|
|
577
|
+
read(): {
|
|
578
|
+
revision: string;
|
|
579
|
+
events: string;
|
|
580
|
+
};
|
|
581
|
+
append(expectedRevision: string, event: string): string | undefined;
|
|
582
|
+
}
|
|
583
|
+
interface CostLedgerOptions {
|
|
584
|
+
costCeilingUsd?: number;
|
|
585
|
+
persistence?: CostLedgerPersistence;
|
|
586
|
+
/** Import already-settled receipts without admitting new paid work. */
|
|
587
|
+
receipts?: readonly CostReceipt[];
|
|
588
|
+
}
|
|
589
|
+
/** Run-wide paid-call admission, durable call state, receipts, and summaries. */
|
|
590
|
+
declare class CostLedger {
|
|
591
|
+
private readonly records;
|
|
592
|
+
private readonly activeCallIds;
|
|
593
|
+
private readonly lateCallIds;
|
|
594
|
+
private completedTasks;
|
|
595
|
+
private revision;
|
|
596
|
+
private costLimitPersisted;
|
|
597
|
+
readonly costCeilingUsd?: number;
|
|
598
|
+
private readonly persistence?;
|
|
599
|
+
constructor(input?: number | CostLedgerOptions);
|
|
600
|
+
runPaidCall<T>(input: RunPaidCallInput<T>): Promise<PaidCallResult<T>>;
|
|
601
|
+
/** Settle a call left pending by a crashed process after reconciling with the provider. */
|
|
602
|
+
reconcile(callId: string, observed: CostReceiptInput, options?: {
|
|
603
|
+
error?: string;
|
|
604
|
+
}): CostReceipt;
|
|
605
|
+
list(filter?: CostLedgerFilter): CostReceipt[];
|
|
606
|
+
summary(filter?: CostLedgerFilter): CostLedgerSummary;
|
|
607
|
+
markCompleted(count?: number): void;
|
|
608
|
+
costPerCompletedTask(): number | null;
|
|
609
|
+
private execute;
|
|
610
|
+
private captureLateOutcome;
|
|
611
|
+
private commitOutcome;
|
|
612
|
+
private captureFailure;
|
|
613
|
+
private commitReceipt;
|
|
614
|
+
private resolveMaximum;
|
|
615
|
+
private hasIncompleteSettledCall;
|
|
616
|
+
private appendRecord;
|
|
617
|
+
private ensureCostLimitPersisted;
|
|
618
|
+
private appendEvent;
|
|
619
|
+
}
|
|
620
|
+
|
|
621
|
+
interface Scenario$1 {
|
|
622
|
+
id: string;
|
|
623
|
+
persona: string;
|
|
624
|
+
label: string;
|
|
625
|
+
thesis: string;
|
|
626
|
+
dimensions: string[];
|
|
627
|
+
turns: Turn[];
|
|
628
|
+
artifactChecks: ArtifactCheck[];
|
|
629
|
+
systemPromptAppend?: string;
|
|
630
|
+
}
|
|
631
|
+
interface Turn {
|
|
632
|
+
user: string;
|
|
633
|
+
expectedBehaviors: string[];
|
|
634
|
+
adversarial?: boolean;
|
|
635
|
+
feedbackType?: 'correction' | 'rejection' | 'vague' | 'contradictory' | 'escalation';
|
|
636
|
+
}
|
|
637
|
+
interface ArtifactCheck {
|
|
638
|
+
type: 'vault_file_exists' | 'vault_file_contains' | 'block_extracted' | 'code_valid' | 'generation_produced' | 'tool_created' | string;
|
|
639
|
+
target: string;
|
|
640
|
+
contains?: string;
|
|
641
|
+
minCount?: number;
|
|
642
|
+
description: string;
|
|
643
|
+
}
|
|
644
|
+
interface TurnResult {
|
|
645
|
+
turnIndex: number;
|
|
646
|
+
userMessage: string;
|
|
647
|
+
agentResponse: string;
|
|
648
|
+
durationMs: number;
|
|
649
|
+
blocksExtracted: {
|
|
650
|
+
type: string;
|
|
651
|
+
title: string;
|
|
652
|
+
}[];
|
|
653
|
+
containsCode: boolean;
|
|
654
|
+
containsToolCall: boolean;
|
|
655
|
+
}
|
|
656
|
+
interface CollectedArtifacts {
|
|
657
|
+
vaultFiles: {
|
|
658
|
+
path: string;
|
|
659
|
+
content: string;
|
|
660
|
+
}[];
|
|
661
|
+
blocksExtracted: {
|
|
662
|
+
type: string;
|
|
663
|
+
fields: Record<string, string>;
|
|
664
|
+
}[];
|
|
665
|
+
codeBlocks: {
|
|
666
|
+
language: string;
|
|
667
|
+
code: string;
|
|
668
|
+
}[];
|
|
669
|
+
toolCalls: string[];
|
|
670
|
+
}
|
|
671
|
+
interface JudgeInput {
|
|
672
|
+
scenario: Scenario$1;
|
|
673
|
+
turns: TurnResult[];
|
|
674
|
+
artifacts: CollectedArtifacts;
|
|
675
|
+
/** Shared ledger for paid built-in judges. Direct calls default to an uncapped ledger. */
|
|
676
|
+
costLedger?: CostLedger;
|
|
677
|
+
costPhase?: string;
|
|
678
|
+
costTags?: Record<string, string>;
|
|
679
|
+
signal?: AbortSignal;
|
|
680
|
+
/** Exact maximum provider attempts configured on the supplied TCloud client. */
|
|
681
|
+
tcloudMaximumAttempts?: number;
|
|
682
|
+
}
|
|
683
|
+
|
|
684
|
+
/**
|
|
685
|
+
* RawProviderSink — first-class persistence for the actual HTTP-level
|
|
686
|
+
* request/response bodies of every LLM provider call.
|
|
687
|
+
*
|
|
688
|
+
* Why this is a separate sink from the structured `LlmSpan`:
|
|
689
|
+
*
|
|
690
|
+
* - `LlmSpan` records the *intent* — model name, messages, output text,
|
|
691
|
+
* usage. It's what dashboards read; it's NOT enough for forensics.
|
|
692
|
+
* - When a downstream consumer reports "the verifier used the wrong route"
|
|
693
|
+
* or "tokens look right but reasoning was missing," the only way to
|
|
694
|
+
* answer is the raw HTTP body. Span fields can lie (a proxy can echo
|
|
695
|
+
* a different `model` value than what actually answered); the raw
|
|
696
|
+
* response is ground truth.
|
|
697
|
+
*
|
|
698
|
+
* Default behaviour: opt-in. Pass `rawSink` to `LlmClientOptions` (or the
|
|
699
|
+
* matrix runner / BuilderSession sets it up automatically) and every
|
|
700
|
+
* request, response, and error is recorded — including retries, with the
|
|
701
|
+
* attempt index attached so a flaky call's full event chain is recoverable.
|
|
702
|
+
*
|
|
703
|
+
* Redaction is enforced at sink time. The default redactor strips
|
|
704
|
+
* `Authorization`, `X-Api-Key`, `X-Auth-Token`, `Cookie` headers and any
|
|
705
|
+
* payload field whose key matches `apiKey | api_key | bearer | password |
|
|
706
|
+
* secret | token` (case-insensitive). Override via the sink constructor or
|
|
707
|
+
* the per-call `redactor`. The `redactedFields` array on the persisted
|
|
708
|
+
* event lets a reviewer see what was stripped without exposing the values.
|
|
709
|
+
*/
|
|
710
|
+
type RawProviderDirection = 'request' | 'response' | 'error';
|
|
711
|
+
interface RawProviderEvent {
|
|
712
|
+
/** Stable id. Generated by the sink if omitted. */
|
|
713
|
+
eventId: string;
|
|
714
|
+
/** Trace context populated by `LlmClient` when the call is wrapped in a span. */
|
|
715
|
+
runId?: string;
|
|
716
|
+
spanId?: string;
|
|
717
|
+
/**
|
|
718
|
+
* Logical provider name. Free-form so callers can use whatever id matches
|
|
719
|
+
* their topology (`'openai'`, `'anthropic'`, `'tangle-router'`, …). When
|
|
720
|
+
* omitted, derived from `baseUrl` in `LlmClientOptions`.
|
|
721
|
+
*/
|
|
722
|
+
provider: string;
|
|
723
|
+
model: string;
|
|
724
|
+
/** Endpoint path, e.g. `'/v1/chat/completions'`. */
|
|
725
|
+
endpoint: string;
|
|
726
|
+
/** Base URL used for the call (already-normalised — no trailing slash). */
|
|
727
|
+
baseUrl: string;
|
|
728
|
+
/** 0-indexed retry attempt. The first attempt is 0; a retried call gets 1, 2, … */
|
|
729
|
+
attemptIndex: number;
|
|
730
|
+
direction: RawProviderDirection;
|
|
731
|
+
/** Unix ms. */
|
|
732
|
+
timestamp: number;
|
|
733
|
+
/** Wall-clock duration of the call leg. Set on `response` and `error` events; null on `request`. */
|
|
734
|
+
durationMs?: number;
|
|
735
|
+
statusCode?: number;
|
|
736
|
+
requestHeaders?: Record<string, string>;
|
|
737
|
+
requestBody?: unknown;
|
|
738
|
+
responseHeaders?: Record<string, string>;
|
|
739
|
+
responseBody?: unknown;
|
|
740
|
+
/** Set on `direction: 'error'` events. */
|
|
741
|
+
errorMessage?: string;
|
|
742
|
+
/** Field paths the redactor stripped from this event ('header:Authorization', 'body.apiKey', …). */
|
|
743
|
+
redactedFields: string[];
|
|
744
|
+
}
|
|
745
|
+
interface RawProviderSinkFilter {
|
|
746
|
+
runId?: string;
|
|
747
|
+
spanId?: string;
|
|
748
|
+
direction?: RawProviderDirection;
|
|
749
|
+
attemptIndex?: number;
|
|
750
|
+
}
|
|
751
|
+
interface RawProviderSink {
|
|
752
|
+
record(event: RawProviderEvent): Promise<void>;
|
|
753
|
+
/** Optional listing — implementations that durably persist (file, db) should support this. */
|
|
754
|
+
list?(filter?: RawProviderSinkFilter): Promise<RawProviderEvent[]>;
|
|
755
|
+
/** Optional teardown for backed implementations. */
|
|
756
|
+
close?(): Promise<void>;
|
|
757
|
+
}
|
|
758
|
+
type ProviderRedactor = (event: RawProviderEvent) => RawProviderEvent;
|
|
759
|
+
|
|
760
|
+
/**
|
|
761
|
+
* LLM client with graceful degrade.
|
|
762
|
+
*
|
|
763
|
+
* OpenAI-compatible `/v1/chat/completions` client with:
|
|
764
|
+
* - Exponential-backoff retry on 429 + 5xx gateway errors (502/503/504).
|
|
765
|
+
* - Retry on transient network errors (fetch failed, AbortError, ECONNRESET).
|
|
766
|
+
* - Graceful json_schema → json_object degrade on 400 with schema-reject body.
|
|
767
|
+
* - Fenced-JSON stripping (```json ... ```) for models that wrap structured output.
|
|
768
|
+
* - Configurable base URL + api key / bearer, works with LiteLLM proxies, OpenAI
|
|
769
|
+
* directly, cli-bridge subscriptions, and any router that speaks the spec.
|
|
770
|
+
*
|
|
771
|
+
* Usage:
|
|
772
|
+
* const { value, result } = await callLlmJson<MyType>(
|
|
773
|
+
* { model: 'gpt-4o', messages: [...], jsonSchema: { name: 'x', schema: {...} } },
|
|
774
|
+
* { baseUrl: 'https://router.tangle.tools/v1', apiKey: process.env.KEY },
|
|
775
|
+
* )
|
|
776
|
+
*
|
|
777
|
+
* This is THE llm-calling seam for agent-eval primitives that need structured
|
|
778
|
+
* output (semantic concept judge, reviewer directives, critic scores). Primitives
|
|
779
|
+
* that need free-form text use `callLlm` and parse output themselves.
|
|
780
|
+
*/
|
|
781
|
+
|
|
782
|
+
interface LlmMessage {
|
|
783
|
+
role: 'system' | 'user' | 'assistant';
|
|
784
|
+
/**
|
|
785
|
+
* Either a plain text content string OR a multimodal content array
|
|
786
|
+
* (text + image_url parts) for vision-capable models.
|
|
787
|
+
*/
|
|
788
|
+
content: string | Array<{
|
|
789
|
+
type: 'text';
|
|
790
|
+
text: string;
|
|
791
|
+
} | {
|
|
792
|
+
type: 'image_url';
|
|
793
|
+
image_url: {
|
|
794
|
+
url: string;
|
|
795
|
+
detail?: 'auto' | 'low' | 'high';
|
|
796
|
+
};
|
|
797
|
+
}>;
|
|
798
|
+
}
|
|
799
|
+
interface LlmCallRequest {
|
|
800
|
+
model: string;
|
|
801
|
+
messages: LlmMessage[];
|
|
802
|
+
/** Optional JSON-mode response format (response_format: json_object). */
|
|
803
|
+
jsonMode?: boolean;
|
|
804
|
+
/** Optional structured output via JSON Schema. Falls back to json_object on 400. */
|
|
805
|
+
jsonSchema?: {
|
|
806
|
+
name: string;
|
|
807
|
+
schema: Record<string, unknown>;
|
|
808
|
+
};
|
|
809
|
+
temperature?: number;
|
|
810
|
+
maxTokens?: number;
|
|
811
|
+
/** Per-call timeout, default 300s. */
|
|
812
|
+
timeoutMs?: number;
|
|
813
|
+
}
|
|
814
|
+
interface LlmUsage {
|
|
815
|
+
promptTokens: number;
|
|
816
|
+
completionTokens: number;
|
|
817
|
+
totalTokens: number;
|
|
818
|
+
/** False when the provider omitted or malformed prompt/completion usage. */
|
|
819
|
+
captured?: boolean;
|
|
820
|
+
/** Proxies populate this when prompt caching is on. */
|
|
821
|
+
cachedPromptTokens?: number;
|
|
822
|
+
}
|
|
823
|
+
interface LlmCallResult {
|
|
824
|
+
/** The text content of the first choice. Empty string if none. */
|
|
825
|
+
content: string;
|
|
826
|
+
usage: LlmUsage;
|
|
827
|
+
/**
|
|
828
|
+
* Cost in USD. Pulled from proxy's `_response_cost` field when present;
|
|
829
|
+
* `null` when neither the proxy nor the caller can derive it.
|
|
830
|
+
*/
|
|
831
|
+
costUsd: number | null;
|
|
832
|
+
/** Model name actually used (echoed from response). */
|
|
833
|
+
model: string;
|
|
834
|
+
/** Wall-clock duration of the HTTP call (last attempt, if retried). */
|
|
835
|
+
durationMs: number;
|
|
836
|
+
/**
|
|
837
|
+
* `finish_reason` echoed from the first choice (`stop`, `length`,
|
|
838
|
+
* `content_filter`, `tool_calls`, ...). `null` when the provider omits it.
|
|
839
|
+
* Exposed so a free-form `callLlm` caller CAN detect a truncated answer
|
|
840
|
+
* (`length`) instead of treating a cut-off completion as complete. Note:
|
|
841
|
+
* `callLlm` does not itself reject on it — acting on this signal is the
|
|
842
|
+
* caller's responsibility (in-repo free-form drivers do not yet enforce it).
|
|
843
|
+
*/
|
|
844
|
+
finishReason?: string | null;
|
|
845
|
+
/**
|
|
846
|
+
* True when `content.trim()` is empty. An empty completion is a silent zero
|
|
847
|
+
* for free-form `callLlm` callers; this flag is the signal a caller can
|
|
848
|
+
* inspect to fail loud rather than proceed on an empty string. `callLlm`
|
|
849
|
+
* surfaces it but does not throw on it.
|
|
850
|
+
*/
|
|
851
|
+
contentEmpty?: boolean;
|
|
852
|
+
/** Raw response body. */
|
|
853
|
+
raw: Record<string, unknown>;
|
|
854
|
+
}
|
|
855
|
+
type LlmCallMetadata = Pick<LlmCallResult, 'usage' | 'costUsd' | 'model' | 'durationMs'>;
|
|
856
|
+
interface LlmClientOptions {
|
|
857
|
+
/** Base URL (without trailing slash). Must end at the `/v1` prefix. */
|
|
858
|
+
baseUrl?: string;
|
|
859
|
+
/** Bearer token — either `apiKey` or `bearer` populates `Authorization: Bearer ...`. */
|
|
860
|
+
apiKey?: string;
|
|
861
|
+
bearer?: string;
|
|
862
|
+
/** Override for the `Authorization` header (e.g. `X-Auth: ...`). Takes precedence over apiKey/bearer. */
|
|
863
|
+
authHeader?: {
|
|
864
|
+
name: string;
|
|
865
|
+
value: string;
|
|
866
|
+
};
|
|
867
|
+
/** Stable provider idempotency key, reused across retries of this logical call. */
|
|
868
|
+
idempotencyKey?: string;
|
|
869
|
+
/** Default timeout in ms. Per-call can override. */
|
|
870
|
+
defaultTimeoutMs?: number;
|
|
871
|
+
/**
|
|
872
|
+
* Caller-supplied abort signal — e.g. a campaign-wide cancel. Linked to
|
|
873
|
+
* each attempt's per-attempt timeout controller, so aborting it cancels
|
|
874
|
+
* the in-flight fetch. A caller abort is FATAL: it is not retried even
|
|
875
|
+
* though an AbortError otherwise matches the transient patterns.
|
|
876
|
+
*/
|
|
877
|
+
signal?: AbortSignal;
|
|
878
|
+
/**
|
|
879
|
+
* Cross-attempt wall-clock budget in ms, measured from the first attempt.
|
|
880
|
+
* Before launching each attempt the loop checks the remaining budget and
|
|
881
|
+
* stops retrying once it is exhausted, rather than waiting the full
|
|
882
|
+
* per-attempt timeout on every retry. Bounds total time independent of
|
|
883
|
+
* total attempts × `timeoutMs`.
|
|
884
|
+
*/
|
|
885
|
+
deadlineMs?: number;
|
|
886
|
+
/** Total provider attempts. Legacy option name; default 3 (1 initial + 2 retries). */
|
|
887
|
+
maxRetries?: number;
|
|
888
|
+
/** Fetch implementation — defaults to global `fetch`. Override for custom transport (e.g. tests). */
|
|
889
|
+
fetch?: typeof fetch;
|
|
890
|
+
/**
|
|
891
|
+
* Optional raw HTTP capture sink. When provided, every request, response,
|
|
892
|
+
* and error (across all retry attempts) is recorded to the sink, with auth
|
|
893
|
+
* headers and credential-shaped body fields redacted by default. This is
|
|
894
|
+
* the layer-1 forensics primitive: structured `LlmSpan`s record intent,
|
|
895
|
+
* raw events record what actually crossed the wire.
|
|
896
|
+
*/
|
|
897
|
+
rawSink?: RawProviderSink;
|
|
898
|
+
/**
|
|
899
|
+
* Logical provider id attached to raw events. When omitted, derived from
|
|
900
|
+
* `baseUrl` via `providerFromBaseUrl`.
|
|
901
|
+
*/
|
|
902
|
+
provider?: string;
|
|
903
|
+
/** Trace context attached to raw events; populated by emitter-aware callers. */
|
|
904
|
+
traceContext?: {
|
|
905
|
+
runId?: string;
|
|
906
|
+
spanId?: string;
|
|
907
|
+
};
|
|
908
|
+
/** Override the redaction strategy for this call. Defaults to `defaultProviderRedactor`. */
|
|
909
|
+
redactor?: ProviderRedactor;
|
|
910
|
+
}
|
|
911
|
+
|
|
912
|
+
/**
|
|
913
|
+
* ChatClient — the single LLM abstraction analysts call.
|
|
914
|
+
*
|
|
915
|
+
* agent-eval already ships an `LlmClient` (OpenAI-compatible, retry,
|
|
916
|
+
* graceful JSON-schema degrade) and judges that talk to `TCloud`. Two
|
|
917
|
+
* mixed patterns force every analyst author to pick a transport, which
|
|
918
|
+
* couples analyst code to runtime concerns (cli-bridge vs router vs
|
|
919
|
+
* sandbox-sdk) it shouldn't know about.
|
|
920
|
+
*
|
|
921
|
+
* `ChatClient` is one interface every analyst takes via `AnalystContext.chat`.
|
|
922
|
+
* The operator decides at the registry boundary which transport binds
|
|
923
|
+
* to it. Analyst code stays transport-agnostic; swapping production
|
|
924
|
+
* (sandbox-sdk) for local dev (cli-bridge) or tests (mock) is a one-
|
|
925
|
+
* line factory call.
|
|
926
|
+
*
|
|
927
|
+
* Designed to coexist: existing `LlmClient` callers and existing
|
|
928
|
+
* `TCloud`-based judges keep working untouched. New analyst code uses
|
|
929
|
+
* `ChatClient`. When old call sites migrate, they pick up budgeting,
|
|
930
|
+
* cancellation, and unified telemetry for free.
|
|
931
|
+
*/
|
|
932
|
+
|
|
933
|
+
/**
|
|
934
|
+
* Unified chat interface. Mirrors LlmCallRequest/Result so the OpenAI-
|
|
935
|
+
* compatible mental model stays. Two methods: a one-shot `chat()` and
|
|
936
|
+
* an `streamChat()` for future agentic loops (not yet exposed).
|
|
937
|
+
*/
|
|
938
|
+
interface ChatClient {
|
|
939
|
+
/** Display name of the bound transport — included in telemetry. */
|
|
940
|
+
readonly transport: ChatTransport;
|
|
941
|
+
/** Default model when caller omits — operators bind this per environment. */
|
|
942
|
+
readonly defaultModel?: string;
|
|
943
|
+
/** Total provider attempts this transport can make for one chat call. */
|
|
944
|
+
readonly maximumAttempts?: number;
|
|
945
|
+
/** Implementations must enforce `req.maxTokens` when it is present. */
|
|
946
|
+
chat(req: ChatRequest, opts?: ChatCallOpts): Promise<ChatResponse>;
|
|
947
|
+
}
|
|
948
|
+
type ChatTransport = 'router' | 'sandbox-sdk' | 'cli-bridge' | 'direct-provider' | 'mock';
|
|
949
|
+
interface ChatRequest extends Omit<LlmCallRequest, 'model'> {
|
|
950
|
+
/** Optional — falls back to ChatClient.defaultModel. */
|
|
951
|
+
model?: string;
|
|
952
|
+
}
|
|
953
|
+
type ChatResponse = LlmCallResult;
|
|
954
|
+
interface ChatCallOpts {
|
|
955
|
+
/** Cancel the in-flight request. */
|
|
956
|
+
signal?: AbortSignal;
|
|
957
|
+
/** Hard USD ceiling for this single call (informational; the underlying transport may not enforce). */
|
|
958
|
+
maxCostUsd?: number;
|
|
959
|
+
/** Correlation tag carried into request headers when the transport allows. */
|
|
960
|
+
correlationId?: string;
|
|
961
|
+
/** Stable provider idempotency key for retries/redrives of one paid call. */
|
|
962
|
+
idempotencyKey?: string;
|
|
963
|
+
}
|
|
964
|
+
type CreateChatClientOpts = RouterTransportOpts | CliBridgeTransportOpts | DirectProviderTransportOpts | SandboxSdkTransportOpts | MockTransportOpts;
|
|
965
|
+
interface BaseTransportOpts {
|
|
966
|
+
defaultModel?: string;
|
|
967
|
+
/** Total provider attempts. Required for opaque transports used in capped runs. */
|
|
968
|
+
maximumAttempts?: number;
|
|
969
|
+
}
|
|
970
|
+
interface RouterTransportOpts extends BaseTransportOpts {
|
|
971
|
+
transport: 'router';
|
|
972
|
+
baseUrl?: string;
|
|
973
|
+
apiKey: string;
|
|
974
|
+
}
|
|
975
|
+
interface CliBridgeTransportOpts extends BaseTransportOpts {
|
|
976
|
+
transport: 'cli-bridge';
|
|
977
|
+
baseUrl?: string;
|
|
978
|
+
bearer?: string;
|
|
979
|
+
}
|
|
980
|
+
interface DirectProviderTransportOpts extends BaseTransportOpts {
|
|
981
|
+
transport: 'direct-provider';
|
|
982
|
+
baseUrl: string;
|
|
983
|
+
apiKey: string;
|
|
984
|
+
}
|
|
985
|
+
/**
|
|
986
|
+
* Sandbox-SDK transport. Provided as a thin pass-through: the caller
|
|
987
|
+
* supplies a callable that mimics LlmClient.chat() against an already-
|
|
988
|
+
* configured Sandbox handle. We don't import the SDK here to keep
|
|
989
|
+
* agent-eval dep-free of @tangle-network/sandbox.
|
|
990
|
+
*/
|
|
991
|
+
interface SandboxSdkTransportOpts extends BaseTransportOpts {
|
|
992
|
+
transport: 'sandbox-sdk';
|
|
993
|
+
chat: (req: ChatRequest, opts?: ChatCallOpts) => Promise<ChatResponse>;
|
|
994
|
+
}
|
|
995
|
+
/**
|
|
996
|
+
* Mock transport for tests. The handler receives the request and returns
|
|
997
|
+
* whatever the test wants. No retries, no JSON-schema degrade.
|
|
998
|
+
*/
|
|
999
|
+
interface MockTransportOpts extends BaseTransportOpts {
|
|
1000
|
+
transport: 'mock';
|
|
1001
|
+
handler: (req: ChatRequest, opts?: ChatCallOpts) => Promise<ChatResponse>;
|
|
1002
|
+
}
|
|
1003
|
+
/**
|
|
1004
|
+
* Build a ChatClient bound to a specific transport. The returned client
|
|
1005
|
+
* is safe to share across analysts in a single registry run.
|
|
1006
|
+
*/
|
|
1007
|
+
declare function createChatClient(opts: CreateChatClientOpts): ChatClient;
|
|
1008
|
+
|
|
1009
|
+
/**
|
|
1010
|
+
* Analyst contract — the missing orchestration layer over agent-eval's
|
|
1011
|
+
* existing analyzers (analyzeTraces, MultiLayerVerifier, RunCritic,
|
|
1012
|
+
* SemanticConceptJudge, JudgeFn, ...).
|
|
1013
|
+
*
|
|
1014
|
+
* Each existing primitive returns its own output shape. The Analyst
|
|
1015
|
+
* contract is the single envelope every primitive lifts into, so a
|
|
1016
|
+
* registry can run N analysts against a run and a single renderer can
|
|
1017
|
+
* compose findings without knowing which analyzer produced them.
|
|
1018
|
+
*
|
|
1019
|
+
* The contract is intentionally domain-agnostic: nothing here knows
|
|
1020
|
+
* about code, voice, RAG, or any particular agent stack. Analysts
|
|
1021
|
+
* declare what INPUT KIND they need (a trace store, an artifact dir,
|
|
1022
|
+
* a RunRecord, a JudgeInput, or `custom`), and the registry routes
|
|
1023
|
+
* the matching input from `AnalystRunInputs`.
|
|
1024
|
+
*/
|
|
1025
|
+
|
|
1026
|
+
/**
|
|
1027
|
+
* Unified envelope every analyst emits. Schema-versioned so renderers
|
|
1028
|
+
* and time-series diffs survive future field additions.
|
|
1029
|
+
*/
|
|
1030
|
+
interface AnalystFinding {
|
|
1031
|
+
schema_version: '1.0.0';
|
|
1032
|
+
/**
|
|
1033
|
+
* Stable hash over identity-defining fields (analyst_id + canonical
|
|
1034
|
+
* claim + area + optional subject). Two findings from two runs that
|
|
1035
|
+
* "are the same finding" share this id — that's what `diffFindings`
|
|
1036
|
+
* uses to compute appeared/disappeared sets across runs.
|
|
1037
|
+
*/
|
|
1038
|
+
finding_id: string;
|
|
1039
|
+
analyst_id: string;
|
|
1040
|
+
produced_at: string;
|
|
1041
|
+
severity: AnalystSeverity;
|
|
1042
|
+
/**
|
|
1043
|
+
* Coarse classification. Renderers group by this. Free-form so
|
|
1044
|
+
* domain-specific analysts can introduce categories without a
|
|
1045
|
+
* schema change ('agent-reasoning', 'verification', 'cost',
|
|
1046
|
+
* 'tool-use', 'safety', 'latency', 'data-quality', ...).
|
|
1047
|
+
*/
|
|
1048
|
+
area: string;
|
|
1049
|
+
claim: string;
|
|
1050
|
+
rationale?: string;
|
|
1051
|
+
evidence_refs: EvidenceRef[];
|
|
1052
|
+
recommended_action?: string;
|
|
1053
|
+
validation_plan?: string;
|
|
1054
|
+
/** 0..1 — the analyst's own confidence. Not calibrated across analysts. */
|
|
1055
|
+
confidence: number;
|
|
1056
|
+
/**
|
|
1057
|
+
* Optional subject the finding is about — leaf id, agent id, request
|
|
1058
|
+
* id. Included in finding_id when present so per-subject findings
|
|
1059
|
+
* diff cleanly across runs.
|
|
1060
|
+
*/
|
|
1061
|
+
subject?: string;
|
|
1062
|
+
/** FIREWALL provenance (docs/learning-flywheel.md): true iff this finding was
|
|
1063
|
+
* lifted from a JUDGE verdict (an acceptance score), not OBSERVED from the
|
|
1064
|
+
* agent's behavior. A judge-derived finding must NEVER be admitted as a
|
|
1065
|
+
* steering input — that is the held-out judge leaking into the loop. Set at
|
|
1066
|
+
* the lift site (createJudgeAdapter); checked by `assertNoJudgeVerdict`.
|
|
1067
|
+
* Provenance, not evidence presence, is the correct discriminator: an
|
|
1068
|
+
* evidence-less trace-analyst observation legitimately steers, while a judge
|
|
1069
|
+
* verdict that happens to cite an artifact must not. */
|
|
1070
|
+
derived_from_judge?: boolean;
|
|
1071
|
+
/** Analyst-private extras; renderers ignore unless they know the analyst. */
|
|
1072
|
+
metadata?: Record<string, unknown>;
|
|
1073
|
+
}
|
|
1074
|
+
type AnalystSeverity = 'critical' | 'high' | 'medium' | 'low' | 'info';
|
|
1075
|
+
interface EvidenceRef {
|
|
1076
|
+
/**
|
|
1077
|
+
* Where the evidence lives. `span` and `event` refer to OTLP trace
|
|
1078
|
+
* elements; `artifact` to a file inside the run's artifact tree;
|
|
1079
|
+
* `finding` to another AnalystFinding (cross-analyst chaining);
|
|
1080
|
+
* `metric` to a named scalar reading the renderer knows how to read.
|
|
1081
|
+
*/
|
|
1082
|
+
kind: 'span' | 'event' | 'artifact' | 'finding' | 'metric';
|
|
1083
|
+
uri: string;
|
|
1084
|
+
excerpt?: string;
|
|
1085
|
+
}
|
|
1086
|
+
/**
|
|
1087
|
+
* The discriminator the registry uses to pass the right input.
|
|
1088
|
+
* `custom` is the escape hatch — analysts that need something else
|
|
1089
|
+
* (e.g. an embedding cache, a partner SDK handle) read it from
|
|
1090
|
+
* `AnalystRunInputs.custom[<analyst id>]`.
|
|
1091
|
+
*/
|
|
1092
|
+
type AnalystInputKind = 'trace-store' | 'artifact-dir' | 'run-record' | 'judge-input' | 'custom';
|
|
1093
|
+
interface AnalystCost {
|
|
1094
|
+
/** `deterministic` analysts MUST NOT call the LLM. */
|
|
1095
|
+
kind: 'deterministic' | 'llm';
|
|
1096
|
+
/** Optional declared upper bound; the registry can enforce a budget. */
|
|
1097
|
+
est_usd_per_run?: number;
|
|
1098
|
+
/** Models the analyst expects to use (informational). */
|
|
1099
|
+
models?: string[];
|
|
1100
|
+
}
|
|
1101
|
+
interface AnalystRequirements {
|
|
1102
|
+
/** Min number of shots / samples the analyst needs to produce signal. */
|
|
1103
|
+
min_shots?: number;
|
|
1104
|
+
/** Capabilities the runtime must supply (e.g. ['network', 'gpu']). */
|
|
1105
|
+
capabilities?: string[];
|
|
1106
|
+
}
|
|
1107
|
+
/**
|
|
1108
|
+
* What's passed to every analyst call. The registry resolves which
|
|
1109
|
+
* field the analyst's `inputKind` selects and asserts it's present.
|
|
1110
|
+
*/
|
|
1111
|
+
interface AnalystRunInputs {
|
|
1112
|
+
traceStore?: TraceAnalysisStore;
|
|
1113
|
+
artifactDir?: string;
|
|
1114
|
+
runRecord?: RunRecord;
|
|
1115
|
+
judgeInput?: JudgeInput;
|
|
1116
|
+
/** Keyed by analyst id; populated by callers that registered custom analysts. */
|
|
1117
|
+
custom?: Record<string, unknown>;
|
|
1118
|
+
}
|
|
1119
|
+
interface AnalystContext {
|
|
1120
|
+
runId: string;
|
|
1121
|
+
/** Stable correlation id so logs from a single registry.run() share a tag. */
|
|
1122
|
+
correlationId: string;
|
|
1123
|
+
/** Wall-clock deadline (epoch ms). Analysts SHOULD honor for graceful cancel. */
|
|
1124
|
+
deadlineMs?: number;
|
|
1125
|
+
/** Per-analyst USD budget. Analysts MAY check before issuing LLM calls. */
|
|
1126
|
+
budgetUsd?: number;
|
|
1127
|
+
/**
|
|
1128
|
+
* Shared chat client. Analysts that call an LLM go through this so
|
|
1129
|
+
* the operator picks transport (sandbox-sdk | router | cli-bridge |
|
|
1130
|
+
* direct-provider | mock) at the registry boundary without touching
|
|
1131
|
+
* analyst code.
|
|
1132
|
+
*/
|
|
1133
|
+
chat?: ChatClient;
|
|
1134
|
+
/**
|
|
1135
|
+
* Findings from a prior run the operator wants the analyst to see as
|
|
1136
|
+
* retrieval context. Kinds that take advantage of cross-run memory
|
|
1137
|
+
* (failure-mode "I saw this cluster last run", knowledge-gap "the wiki
|
|
1138
|
+
* page I asked for is still missing") render these into the actor's
|
|
1139
|
+
* working set. Filtering is the operator's job: pass the slice that
|
|
1140
|
+
* matches the analyst's id, or pass everything and let the kind
|
|
1141
|
+
* filter. Empty / absent means no cross-run context.
|
|
1142
|
+
*/
|
|
1143
|
+
priorFindings?: ReadonlyArray<AnalystFinding>;
|
|
1144
|
+
/** Free-form runtime tags (env, host, op). Findings can echo these into metadata. */
|
|
1145
|
+
tags?: Record<string, string>;
|
|
1146
|
+
/** Logger callback — analysts SHOULD prefer this over console.* for testability. */
|
|
1147
|
+
log?: (msg: string, fields?: Record<string, unknown>) => void;
|
|
1148
|
+
/** Optional abort signal. Analysts SHOULD pass it through to LLM calls. */
|
|
1149
|
+
signal?: AbortSignal;
|
|
1150
|
+
}
|
|
1151
|
+
/**
|
|
1152
|
+
* The minimal contract. Concrete analysts can refine `TInput` so
|
|
1153
|
+
* implementations stay type-safe (e.g. a trace analyst's `TInput` is
|
|
1154
|
+
* `TraceAnalysisStore`); the registry passes the right field from
|
|
1155
|
+
* `AnalystRunInputs` based on `inputKind`.
|
|
1156
|
+
*/
|
|
1157
|
+
interface Analyst<TInput = unknown> {
|
|
1158
|
+
/** Stable identifier — appears in finding_id, telemetry, and registry exclusion lists. */
|
|
1159
|
+
readonly id: string;
|
|
1160
|
+
/** Human-readable. One sentence. */
|
|
1161
|
+
readonly description: string;
|
|
1162
|
+
readonly inputKind: AnalystInputKind;
|
|
1163
|
+
readonly cost: AnalystCost;
|
|
1164
|
+
readonly requires?: AnalystRequirements;
|
|
1165
|
+
/** Bump on breaking changes to claim wording or area so old finding_ids don't collide. */
|
|
1166
|
+
readonly version: string;
|
|
1167
|
+
analyze(input: TInput, ctx: AnalystContext): Promise<AnalystFinding[]>;
|
|
1168
|
+
}
|
|
1169
|
+
interface AnalystRunSummary {
|
|
1170
|
+
analyst_id: string;
|
|
1171
|
+
status: 'ok' | 'skipped' | 'failed';
|
|
1172
|
+
/** Why skipped — missing input, budget exceeded, capability unmet. */
|
|
1173
|
+
reason?: string;
|
|
1174
|
+
findings_count: number;
|
|
1175
|
+
latency_ms: number;
|
|
1176
|
+
cost_usd: number;
|
|
1177
|
+
/** When `status='failed'`: the error class + message, never the full stack. */
|
|
1178
|
+
error?: {
|
|
1179
|
+
class: string;
|
|
1180
|
+
message: string;
|
|
1181
|
+
};
|
|
1182
|
+
}
|
|
1183
|
+
interface AnalystRunResult {
|
|
1184
|
+
run_id: string;
|
|
1185
|
+
correlation_id: string;
|
|
1186
|
+
started_at: string;
|
|
1187
|
+
ended_at: string;
|
|
1188
|
+
findings: AnalystFinding[];
|
|
1189
|
+
per_analyst: AnalystRunSummary[];
|
|
1190
|
+
/** Total LLM cost in USD across all analysts in this registry.run(). */
|
|
1191
|
+
total_cost_usd: number;
|
|
1192
|
+
}
|
|
1193
|
+
/**
|
|
1194
|
+
* Events emitted by `AnalystRegistry.runStream(...)` in real time as
|
|
1195
|
+
* the registry executes. UIs subscribe via `for await (const ev of
|
|
1196
|
+
* registry.runStream(...))`; `registry.run(...)` is a thin collector
|
|
1197
|
+
* over the same stream, so the two surfaces share their invariants.
|
|
1198
|
+
*
|
|
1199
|
+
* Per-finding events are intentionally omitted — analyzers are batch
|
|
1200
|
+
* operations (an Ax actor returns the full `findings:json[]` at the
|
|
1201
|
+
* end of the responder), so streaming inside one analyst would only
|
|
1202
|
+
* emit partial JSON consumers can't render. The kind-completion event
|
|
1203
|
+
* is the right granularity; subscribers wanting per-finding rendering
|
|
1204
|
+
* iterate `event.findings` themselves.
|
|
1205
|
+
*/
|
|
1206
|
+
type AnalystRunEvent = {
|
|
1207
|
+
type: 'run-started';
|
|
1208
|
+
run_id: string;
|
|
1209
|
+
correlation_id: string;
|
|
1210
|
+
started_at: string;
|
|
1211
|
+
/** The ordered list of analyst ids the registry will run. */
|
|
1212
|
+
analyst_ids: ReadonlyArray<string>;
|
|
1213
|
+
} | {
|
|
1214
|
+
type: 'analyst-skipped';
|
|
1215
|
+
summary: AnalystRunSummary;
|
|
1216
|
+
} | {
|
|
1217
|
+
type: 'analyst-started';
|
|
1218
|
+
analyst_id: string;
|
|
1219
|
+
started_at: string;
|
|
1220
|
+
} | {
|
|
1221
|
+
type: 'analyst-completed';
|
|
1222
|
+
/** `summary.status` is `'ok'` for clean completion or `'failed'` for thrown analysts. */
|
|
1223
|
+
summary: AnalystRunSummary;
|
|
1224
|
+
findings: ReadonlyArray<AnalystFinding>;
|
|
1225
|
+
} | {
|
|
1226
|
+
type: 'run-completed';
|
|
1227
|
+
result: AnalystRunResult;
|
|
1228
|
+
};
|
|
1229
|
+
|
|
1230
|
+
type PolicyEditSchemaVersion = 'policy-edit/v1';
|
|
1231
|
+
declare const POLICY_EDIT_AXES: readonly ["carrier", "representation", "budget", "sampling", "output_contract", "tool_contract", "routing", "memory", "agent_profile", "deployment_target"];
|
|
1232
|
+
type PolicyEditAxis = (typeof POLICY_EDIT_AXES)[number];
|
|
1233
|
+
declare const POLICY_EDIT_TARGET_SURFACES: readonly ["prompt", "tool-contract", "runtime-config", "memory", "agent-profile", "code", "deployment"];
|
|
1234
|
+
type PolicyEditTargetSurface = (typeof POLICY_EDIT_TARGET_SURFACES)[number];
|
|
1235
|
+
type PolicyEditRisk = 'low' | 'medium' | 'high' | 'unknown';
|
|
1236
|
+
type PolicyEditGainDirection = 'increase' | 'decrease';
|
|
1237
|
+
type PolicyEditGainUnit = 'absolute' | 'relative' | 'percent' | 'score';
|
|
1238
|
+
interface PolicyEditTarget {
|
|
1239
|
+
surface: PolicyEditTargetSurface;
|
|
1240
|
+
/** Stable path inside the target surface, for example `system-prompt:tools`
|
|
1241
|
+
* or `budget.maxTurns`. */
|
|
1242
|
+
path?: string;
|
|
1243
|
+
/** Optional canonical deployment identity. Store the existing cell, not a
|
|
1244
|
+
* local profile shape. */
|
|
1245
|
+
agentProfileCell?: AgentProfileCell;
|
|
1246
|
+
/** Human label when the path is not enough for a readable audit trail. */
|
|
1247
|
+
label?: string;
|
|
1248
|
+
}
|
|
1249
|
+
type PolicyEditChange = {
|
|
1250
|
+
kind: 'text';
|
|
1251
|
+
mode: 'append' | 'prepend' | 'replace';
|
|
1252
|
+
value: string;
|
|
1253
|
+
/** Required when `mode === 'replace'`; exact match only. */
|
|
1254
|
+
find?: string;
|
|
1255
|
+
} | {
|
|
1256
|
+
kind: 'json';
|
|
1257
|
+
mode: 'set' | 'merge' | 'remove';
|
|
1258
|
+
path: string;
|
|
1259
|
+
value?: AgentProfileJson;
|
|
1260
|
+
};
|
|
1261
|
+
interface PolicyEditExpectedGain {
|
|
1262
|
+
/** Metric this edit is expected to move, e.g. `holdout.composite`. */
|
|
1263
|
+
metric: string;
|
|
1264
|
+
direction: PolicyEditGainDirection;
|
|
1265
|
+
/** Positive magnitude in the metric's native units. */
|
|
1266
|
+
amount: number;
|
|
1267
|
+
unit?: PolicyEditGainUnit;
|
|
1268
|
+
rationale?: string;
|
|
1269
|
+
}
|
|
1270
|
+
interface PolicyEditSource {
|
|
1271
|
+
findingIds: string[];
|
|
1272
|
+
analystIds: string[];
|
|
1273
|
+
evidenceRefs: EvidenceRef[];
|
|
1274
|
+
/** Mirrors `AnalystFinding.derived_from_judge`; admission rejects it. */
|
|
1275
|
+
derivedFromJudge?: boolean;
|
|
1276
|
+
}
|
|
1277
|
+
interface PolicyEdit {
|
|
1278
|
+
schemaVersion: PolicyEditSchemaVersion;
|
|
1279
|
+
editId: string;
|
|
1280
|
+
axis: PolicyEditAxis;
|
|
1281
|
+
target: PolicyEditTarget;
|
|
1282
|
+
change: PolicyEditChange;
|
|
1283
|
+
claim: string;
|
|
1284
|
+
expectedGain: PolicyEditExpectedGain;
|
|
1285
|
+
confidence: number;
|
|
1286
|
+
risk: PolicyEditRisk;
|
|
1287
|
+
source: PolicyEditSource;
|
|
1288
|
+
rationale?: string;
|
|
1289
|
+
validationPlan?: string;
|
|
1290
|
+
metadata?: Record<string, unknown>;
|
|
1291
|
+
}
|
|
1292
|
+
declare const POLICY_EDIT_CANDIDATE_RECORD_SCHEMA: "tangle.policy-edit-candidate.v1";
|
|
1293
|
+
/** JSON-safe attribution carried with a measured candidate and its scores. */
|
|
1294
|
+
interface PolicyEditCandidateRecord {
|
|
1295
|
+
schema: typeof POLICY_EDIT_CANDIDATE_RECORD_SCHEMA;
|
|
1296
|
+
policyEdit: PolicyEdit;
|
|
1297
|
+
}
|
|
1298
|
+
|
|
1299
|
+
/**
|
|
1300
|
+
* Pass A substrate types — `runCampaign` is the one primitive every
|
|
1301
|
+
* eval flow composes from. Three contracts in this file:
|
|
1302
|
+
*
|
|
1303
|
+
* - `Scenario` input set
|
|
1304
|
+
* - `DispatchFn` how to run one scenario → artifact
|
|
1305
|
+
* - `CampaignResult` defined output schema (the contract downstream tools depend on)
|
|
1306
|
+
*
|
|
1307
|
+
* Three more lifted from earlier substrate work (re-exported):
|
|
1308
|
+
*
|
|
1309
|
+
* - `JudgeConfig` pluggable dimensional scorer (0.38)
|
|
1310
|
+
* - `Mutator` optimization-loop surface mutator
|
|
1311
|
+
* - `Gate` promotion gate (`HeldOutGate` and friends adapt to this)
|
|
1312
|
+
*
|
|
1313
|
+
* No new architecture vs 0.38 — Pass A formalizes the shapes so consumers
|
|
1314
|
+
* can build dashboards / CI gates / regression diffs against a stable schema.
|
|
1315
|
+
*/
|
|
1316
|
+
|
|
1317
|
+
/** Stable identifier + kind tag for any scenario. Consumers
|
|
1318
|
+
* extend with their per-domain payload (persona, task, requirement, ...). */
|
|
1319
|
+
interface Scenario {
|
|
1320
|
+
id: string;
|
|
1321
|
+
kind: string;
|
|
1322
|
+
tags?: string[];
|
|
1323
|
+
}
|
|
1324
|
+
/** Context handed to every dispatch invocation. Scoped — every
|
|
1325
|
+
* trace/span carries the cellId, every artifact write lands under the cell's
|
|
1326
|
+
* artifact root, the cost meter accumulates per cell. */
|
|
1327
|
+
interface DispatchContext {
|
|
1328
|
+
cellId: string;
|
|
1329
|
+
rep: number;
|
|
1330
|
+
generation?: number;
|
|
1331
|
+
seed: number;
|
|
1332
|
+
signal: AbortSignal;
|
|
1333
|
+
trace: CampaignTraceWriter;
|
|
1334
|
+
artifacts: CampaignArtifactWriter;
|
|
1335
|
+
cost: CampaignCostMeter;
|
|
1336
|
+
/** Populated when this run is part of a multi-cycle improvement loop. */
|
|
1337
|
+
cycleId?: string;
|
|
1338
|
+
/** Populated when the substrate resumed from a prior cache hit. */
|
|
1339
|
+
resumedFrom?: string;
|
|
1340
|
+
/**
|
|
1341
|
+
* Opaque placement key supplied by `RunCampaignOptions.cellPlacement`.
|
|
1342
|
+
* The substrate forwards it through unchanged; placement-aware Dispatch
|
|
1343
|
+
* implementations (e.g. `httpDispatch` from `/adapters/http`) read it to
|
|
1344
|
+
* route the cell to the right worker / region / sandbox. `undefined`
|
|
1345
|
+
* when no placement strategy is configured.
|
|
1346
|
+
*/
|
|
1347
|
+
placement?: string;
|
|
1348
|
+
}
|
|
1349
|
+
/** One function: scenario + ctx → artifact. Dispatcher chooses
|
|
1350
|
+
* whether to call `runMultishot`, `runLoop`, raw `streamPrompt`, anything. */
|
|
1351
|
+
type DispatchFn<TScenario extends Scenario, TArtifact> = (scenario: TScenario, ctx: DispatchContext) => Promise<TArtifact>;
|
|
1352
|
+
/** One session within a multi-session journey. Dispatch is
|
|
1353
|
+
* invoked once per session in order; state from prior session's artifact
|
|
1354
|
+
* is exposed via `ctx.priorSessionArtifact`. */
|
|
1355
|
+
interface SessionScript<TScenario, TArtifact> {
|
|
1356
|
+
id: string;
|
|
1357
|
+
intent: string;
|
|
1358
|
+
maxTurns?: number;
|
|
1359
|
+
/** When true, knowledge accumulated this session persists to next. */
|
|
1360
|
+
affectsKnowledge?: boolean;
|
|
1361
|
+
/** Optional per-session persona evolution — called after the session
|
|
1362
|
+
* resolves. Returns the persona shape used by the NEXT session. */
|
|
1363
|
+
evolveAfterSession?: (artifact: TArtifact, sessionIndex: number, scenario: TScenario) => TScenario;
|
|
1364
|
+
}
|
|
1365
|
+
interface JudgeDimension {
|
|
1366
|
+
/** JSON field name + score key. */
|
|
1367
|
+
key: string;
|
|
1368
|
+
/** Description shown in the judge's user prompt. */
|
|
1369
|
+
description: string;
|
|
1370
|
+
}
|
|
1371
|
+
/** Pluggable dimensional scorer. `score` is the contract:
|
|
1372
|
+
* given an artifact + scenario, return a `JudgeScore`. This is deliberately a
|
|
1373
|
+
* function, not a fixed LLM-prompt shape — real consumers judge with
|
|
1374
|
+
* ensembles, deterministic checks, or a single LLM call, and the substrate
|
|
1375
|
+
* must not constrain that. The `llmJudge()` helper builds a `score` that does
|
|
1376
|
+
* one LLM call for the common case. `appliesTo` lets a judge run only on
|
|
1377
|
+
* scenarios that match (e.g. a legal-citation judge only on legal scenarios). */
|
|
1378
|
+
interface JudgeConfig<TArtifact, TScenario extends Scenario = Scenario> {
|
|
1379
|
+
name: string;
|
|
1380
|
+
dimensions: JudgeDimension[];
|
|
1381
|
+
/** Stable scoring revision used by campaign resume and verdict caches.
|
|
1382
|
+
* Built-in judges derive this from their prompt, model, and rubric. Custom
|
|
1383
|
+
* judges should set it when closure state can change without changing code. */
|
|
1384
|
+
judgeVersion?: string;
|
|
1385
|
+
/** Score one artifact. Throw on failure — a thrown judge is recorded as a
|
|
1386
|
+
* failed cell, never silently folded into a zero. */
|
|
1387
|
+
score(input: {
|
|
1388
|
+
artifact: TArtifact;
|
|
1389
|
+
scenario: TScenario;
|
|
1390
|
+
signal: AbortSignal;
|
|
1391
|
+
/** Shared run spend account and receipt attribution phase. */
|
|
1392
|
+
costLedger?: CostLedger;
|
|
1393
|
+
costPhase?: string;
|
|
1394
|
+
costTags?: Record<string, string>;
|
|
1395
|
+
}): JudgeScore | Promise<JudgeScore>;
|
|
1396
|
+
appliesTo?: (scenario: TScenario) => boolean;
|
|
1397
|
+
}
|
|
1398
|
+
/** The canonical judge verdict shape — one declaration, shared by campaign
|
|
1399
|
+
* judges and the multishot judge runner (which re-exports this type).
|
|
1400
|
+
*
|
|
1401
|
+
* Scale is PRODUCER-DEFINED: campaign convention is [0,1]; the legacy
|
|
1402
|
+
* multishot runner emits 0-10. Cross-scale comparison must go through
|
|
1403
|
+
* `detectScale` (src/campaign/gates/statistical-heldout.ts, used by
|
|
1404
|
+
* promotion-policy) — never renormalize a producer's values in place, as
|
|
1405
|
+
* downstream thresholds (`composite >= 5` in multishot/matrix.ts, live-soak
|
|
1406
|
+
* `>= 7` gates) key on the producer's native scale. */
|
|
1407
|
+
interface JudgeScore {
|
|
1408
|
+
dimensions: Record<string, number>;
|
|
1409
|
+
composite: number;
|
|
1410
|
+
notes: string;
|
|
1411
|
+
/** Provider metadata for display and diagnostics; accounting uses CostLedger receipts. */
|
|
1412
|
+
llmCall?: LlmCallMetadata;
|
|
1413
|
+
/** Set when the judge itself failed (call error, unparseable output).
|
|
1414
|
+
* `composite`/`dimensions` carry no signal — aggregators MUST exclude
|
|
1415
|
+
* failed scores from means instead of folding them into zeros. */
|
|
1416
|
+
failed?: true;
|
|
1417
|
+
/** Ensemble extras (populated by `ensembleJudge`): max per-dimension
|
|
1418
|
+
* spread across surviving judges — the inter-rater signal. */
|
|
1419
|
+
maxDisagreement?: number;
|
|
1420
|
+
/** Ensemble extras: judge identities whose verdict failed. */
|
|
1421
|
+
failedJudges?: string[];
|
|
1422
|
+
/** Ensemble extras: each surviving judge's per-dimension scores. */
|
|
1423
|
+
perJudge?: Record<string, Record<string, number>>;
|
|
1424
|
+
}
|
|
1425
|
+
/** A tier-4 code surface — a finalized candidate change to the agent's
|
|
1426
|
+
* IMPLEMENTATION, not its prompt. Produced by autoresearch (reads codebase +
|
|
1427
|
+
* trace findings → opens a worktree). `worktreeRef` locates the candidate;
|
|
1428
|
+
* the exact commits, tree, and binary-patch digest identify it. See the
|
|
1429
|
+
* improvement-tier table in `docs/design/loop-taxonomy.md`. */
|
|
1430
|
+
interface CodeSurface {
|
|
1431
|
+
readonly kind: 'code';
|
|
1432
|
+
/** Worktree path or git ref holding the candidate code change. This is a
|
|
1433
|
+
* mutable locator and is deliberately excluded from content hashes. */
|
|
1434
|
+
readonly worktreeRef: string;
|
|
1435
|
+
/** Human-readable ref the worktree was forked from. Not identity-bearing. */
|
|
1436
|
+
readonly baseRef: string;
|
|
1437
|
+
/** Exact commit the candidate was forked from. */
|
|
1438
|
+
readonly baseCommit: string;
|
|
1439
|
+
/** Exact tree object for `baseCommit`. */
|
|
1440
|
+
readonly baseTree: string;
|
|
1441
|
+
/** Exact finalized candidate commit. */
|
|
1442
|
+
readonly candidateCommit: string;
|
|
1443
|
+
/** Exact tree object for `candidateCommit`. */
|
|
1444
|
+
readonly candidateTree: string;
|
|
1445
|
+
/** Identity of the exact patch artifact. The deployable candidate bundle
|
|
1446
|
+
* carries the same descriptor plus its base64-encoded content. */
|
|
1447
|
+
readonly patch: {
|
|
1448
|
+
readonly format: 'git-diff-binary';
|
|
1449
|
+
readonly sha256: `sha256:${string}`;
|
|
1450
|
+
readonly byteLength: number;
|
|
1451
|
+
};
|
|
1452
|
+
/** Human summary of what changed — rendered into the auto-PR body. */
|
|
1453
|
+
readonly summary?: string;
|
|
1454
|
+
}
|
|
1455
|
+
/** The mutable surface a proposer changes. Tiers (see
|
|
1456
|
+
* `docs/design/loop-taxonomy.md`):
|
|
1457
|
+
* - `string` — tiers 1-2: system-prompt addendum / serialized tool
|
|
1458
|
+
* config. Cheap, reversible, text-diffable.
|
|
1459
|
+
* - `CodeSurface` — tier 4: an implementation change behind a worktree ref.
|
|
1460
|
+
* Tier 3 (knowledge) is owned by agent-knowledge and rides its own adapter,
|
|
1461
|
+
* not this type. */
|
|
1462
|
+
type MutableSurface = string | CodeSurface;
|
|
1463
|
+
/** A proposer output carrying the surface AND the WHY behind
|
|
1464
|
+
* it. Reflective proposers (`gepaProposer`) parse a `{label, rationale, payload}`
|
|
1465
|
+
* from the model; without this wrapper the loop keeps only `payload` and the
|
|
1466
|
+
* rationale that motivated the change is lost — the candidate becomes
|
|
1467
|
+
* unattributable. `propose()` may return either bare `MutableSurface`s (cheap
|
|
1468
|
+
* blind mutators) or these (reflective proposers); the loop normalizes both. */
|
|
1469
|
+
interface ProposedCandidate {
|
|
1470
|
+
surface: MutableSurface;
|
|
1471
|
+
/** Short human label for the change (≤ 40 chars typical). */
|
|
1472
|
+
label: string;
|
|
1473
|
+
/** Why this change was proposed — which failure it targets, which
|
|
1474
|
+
* primitive it used. Survives to `GenerationCandidate.rationale` and the
|
|
1475
|
+
* emitted provenance record. */
|
|
1476
|
+
rationale: string;
|
|
1477
|
+
/** Structured, JSON-safe cause for this exact candidate when the proposer
|
|
1478
|
+
* can provide one. Policy edits retain the full validated edit here. */
|
|
1479
|
+
candidateRecord?: PolicyEditCandidateRecord;
|
|
1480
|
+
}
|
|
1481
|
+
/** A non-dominated parent on the GEPA Pareto frontier — a
|
|
1482
|
+
* surface that, across the per-scenario objective vectors, no other tried
|
|
1483
|
+
* surface beats on every scenario. A candidate worse on the mean composite
|
|
1484
|
+
* but uniquely best on one hard scenario is non-dominated and survives here;
|
|
1485
|
+
* the composite-best ranking would discard the lesson it carries. The loop
|
|
1486
|
+
* computes the frontier across ALL generations and hands it to the proposer so
|
|
1487
|
+
* a reflective proposer can combine complementary lessons (GEPA, Agrawal et
|
|
1488
|
+
* al., arXiv:2507.19457). See `pareto.ts` (`paretoFrontier`). */
|
|
1489
|
+
interface ParetoParent {
|
|
1490
|
+
surface: MutableSurface;
|
|
1491
|
+
surfaceHash: string;
|
|
1492
|
+
/** The objective vector: per-scenario composite (higher is better). The
|
|
1493
|
+
* axes the frontier is computed over. */
|
|
1494
|
+
objectives: Record<string, number>;
|
|
1495
|
+
/** Mean composite across the objective scenarios — the scalar summary used
|
|
1496
|
+
* for ordering + display, NOT for dominance. */
|
|
1497
|
+
composite: number;
|
|
1498
|
+
/** Generation that produced this surface (`-1` for the baseline). */
|
|
1499
|
+
generation: number;
|
|
1500
|
+
label?: string;
|
|
1501
|
+
rationale?: string;
|
|
1502
|
+
}
|
|
1503
|
+
/** Exact measured state for the surface an optimizer is learning from.
|
|
1504
|
+
* Unlike a model-authored expected gain, every value here comes from a
|
|
1505
|
+
* completed campaign over the designed denominator. */
|
|
1506
|
+
interface ScoredSurfaceOutcome {
|
|
1507
|
+
/** Optimization/search evidence only. Held-out results must never flow back
|
|
1508
|
+
* into a proposer through this type. */
|
|
1509
|
+
split: 'search';
|
|
1510
|
+
/** Generation that actually measured this surface (`-1` for the baseline). */
|
|
1511
|
+
generation: number;
|
|
1512
|
+
surfaceHash: string;
|
|
1513
|
+
composite: number;
|
|
1514
|
+
dimensions: Record<string, number>;
|
|
1515
|
+
scenarios: Array<{
|
|
1516
|
+
scenarioId: string;
|
|
1517
|
+
composite: number;
|
|
1518
|
+
notes?: string;
|
|
1519
|
+
}>;
|
|
1520
|
+
coverage: {
|
|
1521
|
+
expectedCells: number;
|
|
1522
|
+
scorableCells: number;
|
|
1523
|
+
};
|
|
1524
|
+
}
|
|
1525
|
+
/** Stateless surface mutation — given findings + current
|
|
1526
|
+
* surface, return N candidate surfaces. Pure transform, no generation
|
|
1527
|
+
* awareness. Reflective-mutation and `AxGEPA` mutators conform. Wrapped by
|
|
1528
|
+
* `evolutionaryProposer` to become a `SurfaceProposer`. */
|
|
1529
|
+
interface Mutator<TFindings = unknown> {
|
|
1530
|
+
kind: string;
|
|
1531
|
+
mutate(args: {
|
|
1532
|
+
findings: TFindings[];
|
|
1533
|
+
currentSurface: MutableSurface;
|
|
1534
|
+
populationSize: number;
|
|
1535
|
+
signal: AbortSignal;
|
|
1536
|
+
}): Promise<Array<MutableSurface | ProposedCandidate>>;
|
|
1537
|
+
}
|
|
1538
|
+
/** Everything a proposer may read to plan the next
|
|
1539
|
+
* batch of candidates. The first six fields are always present; the rest are
|
|
1540
|
+
* optional context the loop supplies when available, so cheap proposers
|
|
1541
|
+
* (`evolutionaryProposer`) can ignore them while a code-tier agentic generator
|
|
1542
|
+
* consumes the report + dataset to drive a coding harness.
|
|
1543
|
+
* See `docs/campaign-proposers.md`. */
|
|
1544
|
+
interface ProposeContext<TFindings = unknown> {
|
|
1545
|
+
currentSurface: MutableSurface;
|
|
1546
|
+
history: GenerationRecord[];
|
|
1547
|
+
findings: TFindings[];
|
|
1548
|
+
/** BREADTH: how many candidate surfaces to return this generation. */
|
|
1549
|
+
populationSize: number;
|
|
1550
|
+
generation: number;
|
|
1551
|
+
signal: AbortSignal;
|
|
1552
|
+
/** Measured baseline for this optimization run. `runOptimization` always
|
|
1553
|
+
* supplies it; optional for standalone proposer callers. */
|
|
1554
|
+
baselineOutcome?: ScoredSurfaceOutcome;
|
|
1555
|
+
/** Measured result for `currentSurface`, the complete global incumbent every
|
|
1556
|
+
* new candidate mutates. `runOptimization` always supplies it. */
|
|
1557
|
+
incumbentOutcome?: ScoredSurfaceOutcome;
|
|
1558
|
+
/** Optional analysis report produced before proposal. Opaque to the substrate:
|
|
1559
|
+
* the proposer that consumes it owns the shape. */
|
|
1560
|
+
report?: unknown;
|
|
1561
|
+
/** Handle to all captured data — the proposer samples traces / artifacts /
|
|
1562
|
+
* rewards here to ground its proposals. */
|
|
1563
|
+
dataset?: LabeledScenarioStore;
|
|
1564
|
+
/** DEPTH: max iterations the agentic generator may take per candidate.
|
|
1565
|
+
* 1 = single-shot; >1 = it may iterate on its own change before handing it
|
|
1566
|
+
* back to be measured. */
|
|
1567
|
+
maxImprovementShots?: number;
|
|
1568
|
+
/** GEPA Pareto frontier across ALL generations so far — the non-dominated
|
|
1569
|
+
* surfaces by per-scenario objective vector. Empty/absent on generation 0
|
|
1570
|
+
* (only the baseline is scored). A reflective proposer combines the
|
|
1571
|
+
* complementary lessons of these parents (each excels on different
|
|
1572
|
+
* scenarios) into a merged candidate. Proposers doing pure single-parent
|
|
1573
|
+
* reflection may ignore it. See {@link ParetoParent}. */
|
|
1574
|
+
paretoParents?: ParetoParent[];
|
|
1575
|
+
/** Shared run spend account and receipt attribution phase. */
|
|
1576
|
+
costLedger?: CostLedger;
|
|
1577
|
+
costPhase?: string;
|
|
1578
|
+
/** FIREWALL (non-negotiable): the held-out judge is write-only — its verdicts
|
|
1579
|
+
* score the chosen output and gate promotion, and are NEVER an input to
|
|
1580
|
+
* proposal/steering (else the optimizer games the acceptance axis = an
|
|
1581
|
+
* oracle). This `never`-typed field makes that a compile-time tripwire: a
|
|
1582
|
+
* proposer that tries to thread judge verdicts into the proposal will not type.
|
|
1583
|
+
* Steering may consume TRACE-OBSERVABLE signals (what the agent did) via
|
|
1584
|
+
* `findings`/`report`; it may NOT consume the judge's held-out verdict. */
|
|
1585
|
+
judgeScores?: never;
|
|
1586
|
+
}
|
|
1587
|
+
/** A surface-improvement strategy. Given the current best
|
|
1588
|
+
* surface, the history of what's been tried + scored, and any external
|
|
1589
|
+
* findings, propose the next batch of candidate surfaces to measure.
|
|
1590
|
+
* Optionally decide to stop early.
|
|
1591
|
+
*
|
|
1592
|
+
* The evolutionary mutator (`evolutionaryProposer`, here) and agent-runtime's
|
|
1593
|
+
* reflective / agentic generators both conform. They are proposers for the
|
|
1594
|
+
* SAME loop, not separate loops. The loop body (`runOptimization`) and the
|
|
1595
|
+
* gated promotion shell (`runImprovementLoop`) are proposer-agnostic.
|
|
1596
|
+
*
|
|
1597
|
+
* This is THE optimization proposer — every optimizer is a factory
|
|
1598
|
+
* `xProposer(opts): SurfaceProposer` (`evolutionaryProposer`, `aceProposer`,
|
|
1599
|
+
* `gepaProposer`, `skillOptProposer`, `traceAnalystProposer`, `haloProposer`,
|
|
1600
|
+
* `memoryCurationProposer`, `fapoProposer`), all exported from `/campaign` and
|
|
1601
|
+
* drivable by `selfImprove({ proposer })`. Not to be confused with the
|
|
1602
|
+
* behavior-fuzzing `MutationProposer` (`fuzz/types`), a scenario generator for
|
|
1603
|
+
* a different loop.
|
|
1604
|
+
*/
|
|
1605
|
+
interface SurfaceProposer<TFindings = unknown> {
|
|
1606
|
+
kind: string;
|
|
1607
|
+
/** Plan: propose N candidate surfaces for the next generation. A proposer
|
|
1608
|
+
* may return bare `MutableSurface`s or `ProposedCandidate`s that carry the
|
|
1609
|
+
* `{label, rationale}` motivating the change — the loop threads the
|
|
1610
|
+
* rationale into `GenerationCandidate` and the emitted provenance. */
|
|
1611
|
+
propose(ctx: ProposeContext<TFindings>): Promise<Array<MutableSurface | ProposedCandidate>>;
|
|
1612
|
+
/** Decide: stop early when the proposer judges the search converged or
|
|
1613
|
+
* exhausted. Default (omitted) runs all `maxGenerations`. */
|
|
1614
|
+
decide?(args: {
|
|
1615
|
+
history: GenerationRecord[];
|
|
1616
|
+
}): {
|
|
1617
|
+
stop: boolean;
|
|
1618
|
+
reason?: string;
|
|
1619
|
+
};
|
|
1620
|
+
}
|
|
1621
|
+
/** Optional vocabulary alias. The loop is the optimizer; this object is the
|
|
1622
|
+
* proposer inside that loop. */
|
|
1623
|
+
type OptimizationProposer<TFindings = unknown> = SurfaceProposer<TFindings>;
|
|
1624
|
+
interface OptimizerConfigBase {
|
|
1625
|
+
populationSize: number;
|
|
1626
|
+
maxGenerations: number;
|
|
1627
|
+
surfaceExtractor: (profile: unknown) => MutableSurface;
|
|
1628
|
+
}
|
|
1629
|
+
interface OptimizerConfig extends OptimizerConfigBase {
|
|
1630
|
+
proposer: SurfaceProposer;
|
|
1631
|
+
}
|
|
1632
|
+
/** Five-valued verdict taxonomy (MOSS-paper alignment). */
|
|
1633
|
+
type GateDecision = 'ship' | 'hold' | 'need_more_work' | 'model_ceiling' | 'arch_ceiling';
|
|
1634
|
+
interface GateContext<TArtifact, TScenario extends Scenario> {
|
|
1635
|
+
candidateArtifacts: Map<string, TArtifact>;
|
|
1636
|
+
baselineArtifacts?: Map<string, TArtifact>;
|
|
1637
|
+
/** Candidate (winner) judge scores, keyed by cellId. */
|
|
1638
|
+
judgeScores: Map<string, Record<string, JudgeScore>>;
|
|
1639
|
+
/** Baseline judge scores, keyed by cellId. SEPARATE from `judgeScores` —
|
|
1640
|
+
* baseline + candidate share cellIds (same scenarios), so a single map
|
|
1641
|
+
* cannot represent both. A gate computing a holdout delta MUST read
|
|
1642
|
+
* candidate from `judgeScores` and baseline from here. */
|
|
1643
|
+
baselineJudgeScores?: Map<string, Record<string, JudgeScore>>;
|
|
1644
|
+
/** Neutralized-arm judge scores, keyed by cellId — the winner surface with its
|
|
1645
|
+
* content footprint-matched-blanked (via a `neutralize` fn). Same scenarios as
|
|
1646
|
+
* `judgeScores`. Present ONLY when `runImprovementLoop` was given a `neutralize`
|
|
1647
|
+
* function. A placebo gate (`neutralizationGate`) compares this arm's lift
|
|
1648
|
+
* against the candidate's to reject decorative wins (lift from footprint, not
|
|
1649
|
+
* content). Undefined otherwise. */
|
|
1650
|
+
neutralizedJudgeScores?: Map<string, Record<string, JudgeScore>>;
|
|
1651
|
+
/** Neutralized-arm artifacts, keyed by cellId. Present alongside
|
|
1652
|
+
* `neutralizedJudgeScores`. */
|
|
1653
|
+
neutralizedArtifacts?: Map<string, TArtifact>;
|
|
1654
|
+
scenarios: TScenario[];
|
|
1655
|
+
cost: {
|
|
1656
|
+
candidate: number;
|
|
1657
|
+
baseline: number;
|
|
1658
|
+
};
|
|
1659
|
+
/** Shared run spend account and receipt attribution phase. */
|
|
1660
|
+
costLedger?: CostLedger;
|
|
1661
|
+
costPhase?: string;
|
|
1662
|
+
signal: AbortSignal;
|
|
1663
|
+
}
|
|
1664
|
+
interface GateResult {
|
|
1665
|
+
decision: GateDecision;
|
|
1666
|
+
reasons: string[];
|
|
1667
|
+
contributingGates: Array<{
|
|
1668
|
+
name: string;
|
|
1669
|
+
passed: boolean;
|
|
1670
|
+
detail: unknown;
|
|
1671
|
+
}>;
|
|
1672
|
+
delta?: number;
|
|
1673
|
+
}
|
|
1674
|
+
/** Composable promotion gate. */
|
|
1675
|
+
interface Gate<TArtifact = unknown, TScenario extends Scenario = Scenario> {
|
|
1676
|
+
name: string;
|
|
1677
|
+
decide(ctx: GateContext<TArtifact, TScenario>): Promise<GateResult>;
|
|
1678
|
+
}
|
|
1679
|
+
/** Scoped trace writer handed to each dispatch — every span
|
|
1680
|
+
* auto-tagged with the cellId so traces filter cleanly. */
|
|
1681
|
+
interface CampaignTraceWriter {
|
|
1682
|
+
span(name: string, attributes?: Record<string, unknown>): TraceSpan;
|
|
1683
|
+
flush(): Promise<void>;
|
|
1684
|
+
}
|
|
1685
|
+
interface TraceSpan {
|
|
1686
|
+
end(attributes?: Record<string, unknown>): void;
|
|
1687
|
+
setAttribute(key: string, value: unknown): void;
|
|
1688
|
+
}
|
|
1689
|
+
/** Scoped artifact writer — `write(path, content)` lands under
|
|
1690
|
+
* `<runDir>/<cellId>/<path>`. */
|
|
1691
|
+
interface CampaignArtifactWriter {
|
|
1692
|
+
write(path: string, content: string | Uint8Array): Promise<string>;
|
|
1693
|
+
writeJson(path: string, value: unknown): Promise<string>;
|
|
1694
|
+
}
|
|
1695
|
+
/** Token usage accumulated for a cell. Aliased to the canonical `RunTokenUsage`
|
|
1696
|
+
* (run-record.ts, same package) so a cell maps onto a `RunRecord` for the
|
|
1697
|
+
* backend-integrity guard with ONE source of truth — a field added to
|
|
1698
|
+
* `RunTokenUsage` is a compile error here, not a silent drift. */
|
|
1699
|
+
type CampaignTokenUsage = RunTokenUsage;
|
|
1700
|
+
/** Cell-scoped paid-call entry point. The dispatch places every paid operation
|
|
1701
|
+
* inside `runPaidCall`; the returned provider result supplies one receipt with
|
|
1702
|
+
* cost, tokens, and resolved model. Calls made outside this method are not
|
|
1703
|
+
* admitted or captured. */
|
|
1704
|
+
interface CampaignCostMeter {
|
|
1705
|
+
/** The only paid-call path. Returns a typed result; callers must inspect it. */
|
|
1706
|
+
runPaidCall<T>(input: Omit<RunPaidCallInput<T>, 'channel' | 'phase' | 'tags'> & {
|
|
1707
|
+
channel?: CostChannel;
|
|
1708
|
+
}): Promise<PaidCallResult<T>>;
|
|
1709
|
+
}
|
|
1710
|
+
/** Source tag — required on every store write. Used by the
|
|
1711
|
+
* default training-source filter (production-trace samples NOT used as
|
|
1712
|
+
* training scenarios unless explicitly opted in). */
|
|
1713
|
+
type LabeledScenarioSource = 'production-trace' | 'eval-run' | 'manual' | 'red-team' | 'synthetic';
|
|
1714
|
+
type RedactionStatus = 'raw' | 'redacted-pii' | 'redacted-secrets' | 'fully-redacted';
|
|
1715
|
+
/** How much a label can be trusted to evaluate against — the gold-admission
|
|
1716
|
+
* gate. Strictly ordered: a record qualifies for a `minTrust` filter when its
|
|
1717
|
+
* trust rank is >= the requested rank.
|
|
1718
|
+
*
|
|
1719
|
+
* - `unverified` — label is a heuristic (e.g. raw outcome success/fail).
|
|
1720
|
+
* Fine as corpus; MUST NOT enter a gold set that lift
|
|
1721
|
+
* numbers are computed against.
|
|
1722
|
+
* - `verified-signal` — an external signal confirmed the outcome (PR merged,
|
|
1723
|
+
* tests green, user did not retry, downstream check).
|
|
1724
|
+
* - `human-rated` — a human explicitly rated or corrected the artifact.
|
|
1725
|
+
*
|
|
1726
|
+
* Absent on a write ⇒ treated as `unverified` (fail-closed: a writer must
|
|
1727
|
+
* explicitly assert trust to make a record gold-eligible — it never happens
|
|
1728
|
+
* by accident). */
|
|
1729
|
+
type LabelTrust = 'unverified' | 'verified-signal' | 'human-rated';
|
|
1730
|
+
/** Required-provenance write. The store rejects writes that
|
|
1731
|
+
* lack provenance — a default-on flywheel without provenance is the
|
|
1732
|
+
* data-poisoning vector flagged in the alignment review. */
|
|
1733
|
+
interface LabeledScenarioWrite<TScenario extends Scenario = Scenario, TArtifact = unknown> {
|
|
1734
|
+
scenario: TScenario;
|
|
1735
|
+
artifact: TArtifact;
|
|
1736
|
+
judgeScores: Record<string, JudgeScore>;
|
|
1737
|
+
source: LabeledScenarioSource;
|
|
1738
|
+
sourceVersionHash: string;
|
|
1739
|
+
capturedAt: string;
|
|
1740
|
+
redactionStatus: RedactionStatus;
|
|
1741
|
+
/** Gold-admission trust tier. Absent ⇒ `unverified` (fail-closed): the
|
|
1742
|
+
* record is corpus, never gold. A writer must explicitly assert
|
|
1743
|
+
* `verified-signal` or `human-rated` to make it eligible for a gold
|
|
1744
|
+
* sample. See {@link LabelTrust}. */
|
|
1745
|
+
labelTrust?: LabelTrust;
|
|
1746
|
+
/** Optional per-source rate-limit bucket key (e.g., the tenant id). */
|
|
1747
|
+
rateLimitBucket?: string;
|
|
1748
|
+
}
|
|
1749
|
+
interface LabeledScenarioRecord<TScenario extends Scenario = Scenario, TArtifact = unknown> extends LabeledScenarioWrite<TScenario, TArtifact> {
|
|
1750
|
+
/** Stable hash of (scenario.id, source, capturedAt, sourceVersionHash). */
|
|
1751
|
+
recordHash: string;
|
|
1752
|
+
/** Substrate-assigned split — train if captured before the campaign's
|
|
1753
|
+
* `temporalCutoff`, test if after. Explicit override allowed via filter. */
|
|
1754
|
+
split: 'train' | 'test';
|
|
1755
|
+
}
|
|
1756
|
+
interface LabeledScenarioSampleArgs {
|
|
1757
|
+
count: number;
|
|
1758
|
+
/** REQUIRED — substrate refuses to sample without an explicit split. */
|
|
1759
|
+
split: 'train' | 'test';
|
|
1760
|
+
/** REQUIRED — only records captured before this timestamp are returned.
|
|
1761
|
+
* Enforces temporal split discipline (test scenarios captured AFTER train
|
|
1762
|
+
* cannot enter the training pool). */
|
|
1763
|
+
capturedBefore: string;
|
|
1764
|
+
filter?: {
|
|
1765
|
+
kind?: string;
|
|
1766
|
+
source?: LabeledScenarioSource | LabeledScenarioSource[];
|
|
1767
|
+
minComposite?: number;
|
|
1768
|
+
maxComposite?: number;
|
|
1769
|
+
/** Gold gate: only records whose trust rank is >= this tier are
|
|
1770
|
+
* returned. `sample({ split: 'test', minTrust: 'verified-signal' })` is
|
|
1771
|
+
* the canonical "give me the gold set" call. Absent ⇒ no trust gate
|
|
1772
|
+
* (corpus-level read). */
|
|
1773
|
+
minTrust?: LabelTrust;
|
|
1774
|
+
};
|
|
1775
|
+
}
|
|
1776
|
+
interface LabeledScenarioStore {
|
|
1777
|
+
observe(write: LabeledScenarioWrite): Promise<void>;
|
|
1778
|
+
sample(args: LabeledScenarioSampleArgs): Promise<LabeledScenarioRecord[]>;
|
|
1779
|
+
size(): Promise<{
|
|
1780
|
+
train: number;
|
|
1781
|
+
test: number;
|
|
1782
|
+
bySource: Record<string, number>;
|
|
1783
|
+
/** Count by trust tier — tells the flywheel how much gold it has
|
|
1784
|
+
* accumulated vs. raw corpus. */
|
|
1785
|
+
byTrust: Record<LabelTrust, number>;
|
|
1786
|
+
}>;
|
|
1787
|
+
}
|
|
1788
|
+
interface CampaignCellResult<TArtifact> {
|
|
1789
|
+
/** Manifest that produced this cell. Resumability refuses to reuse a cell
|
|
1790
|
+
* whose manifest differs from the current run. */
|
|
1791
|
+
manifestHash?: string;
|
|
1792
|
+
cellId: string;
|
|
1793
|
+
scenarioId: string;
|
|
1794
|
+
rep: number;
|
|
1795
|
+
generation?: number;
|
|
1796
|
+
artifact: TArtifact;
|
|
1797
|
+
judgeScores: Record<string, JudgeScore>;
|
|
1798
|
+
costUsd: number;
|
|
1799
|
+
/** True when at least one priced receipt used the model table instead of a provider bill. */
|
|
1800
|
+
costEstimated?: boolean;
|
|
1801
|
+
/** Exact durable receipts required to reuse this cached result. */
|
|
1802
|
+
costCallIds?: string[];
|
|
1803
|
+
/** Agent-call token usage committed by `ctx.cost.runPaidCall`.
|
|
1804
|
+
* `{ input: 0, output: 0 }` when no paid agent call was recorded. */
|
|
1805
|
+
tokenUsage: CampaignTokenUsage;
|
|
1806
|
+
/** Concrete model from the latest committed agent receipt. Consumed by
|
|
1807
|
+
* `buildRunRecord` to pin the model when the declared profile uses a
|
|
1808
|
+
* runtime-resolved sentinel. */
|
|
1809
|
+
resolvedModel?: string;
|
|
1810
|
+
durationMs: number;
|
|
1811
|
+
seed: number;
|
|
1812
|
+
cached: boolean;
|
|
1813
|
+
error?: string;
|
|
1814
|
+
}
|
|
1815
|
+
interface JudgeAggregate {
|
|
1816
|
+
mean: number;
|
|
1817
|
+
stdev: number;
|
|
1818
|
+
ci95: [number, number];
|
|
1819
|
+
n: number;
|
|
1820
|
+
}
|
|
1821
|
+
interface ScenarioAggregate {
|
|
1822
|
+
meanComposite: number;
|
|
1823
|
+
ci95: [number, number];
|
|
1824
|
+
n: number;
|
|
1825
|
+
}
|
|
1826
|
+
interface GenerationRecord {
|
|
1827
|
+
generationIndex: number;
|
|
1828
|
+
candidates: GenerationCandidate[];
|
|
1829
|
+
promoted: string[];
|
|
1830
|
+
}
|
|
1831
|
+
/** One scored candidate surface in a generation. `dimensions` + `scenarios`
|
|
1832
|
+
* let a reflective proposer ground its next proposal on WHICH
|
|
1833
|
+
* dimensions the candidate is weakest on and WHICH scenarios it best/worst
|
|
1834
|
+
* handled — the evidence a blind `Mutator` cannot see. */
|
|
1835
|
+
interface GenerationCandidate {
|
|
1836
|
+
surfaceHash: string;
|
|
1837
|
+
composite: number;
|
|
1838
|
+
ci95: [number, number];
|
|
1839
|
+
/** Exact surface this candidate mutated. */
|
|
1840
|
+
parentSurfaceHash?: string;
|
|
1841
|
+
/** Measured search-split composite of the exact parent surface. */
|
|
1842
|
+
parentComposite?: number;
|
|
1843
|
+
/** Candidate composite minus its parent's composite. Present only when the
|
|
1844
|
+
* candidate completed the designed denominator. */
|
|
1845
|
+
observedDeltaFromParent?: number;
|
|
1846
|
+
/** Whether this candidate had a scorable result for every designed campaign
|
|
1847
|
+
* cell and was therefore eligible for ranking, promotion, and Pareto
|
|
1848
|
+
* selection. Older externally-authored records may omit this field; loop
|
|
1849
|
+
* records always populate it. */
|
|
1850
|
+
eligibleForPromotion?: boolean;
|
|
1851
|
+
/** Exact denominator receipt for selection eligibility. Scores stay
|
|
1852
|
+
* descriptive: an incomplete candidate is retained with its observed score
|
|
1853
|
+
* and errors instead of receiving an invented penalty. */
|
|
1854
|
+
coverage?: {
|
|
1855
|
+
expectedCells: number;
|
|
1856
|
+
scorableCells: number;
|
|
1857
|
+
unscorableCells: Array<{
|
|
1858
|
+
cellId: string;
|
|
1859
|
+
reason: string;
|
|
1860
|
+
}>;
|
|
1861
|
+
};
|
|
1862
|
+
/** Mean score per judge dimension across all cells (scenarios × reps ×
|
|
1863
|
+
* judges that reported the dimension). */
|
|
1864
|
+
dimensions: Record<string, number>;
|
|
1865
|
+
/** Per-scenario composite (mean over reps + judges), plus the judge's
|
|
1866
|
+
* free-form `notes` for that scenario — the "why it scored low" evidence a
|
|
1867
|
+
* reflective proposer grounds its next edit on. Keep `notes` GENERALIZABLE
|
|
1868
|
+
* (which checks/lines/dimensions failed and how), NOT case-specific ground
|
|
1869
|
+
* truth: leaking expected answers into the prompt is memorization, and the
|
|
1870
|
+
* held-out gate would reject it anyway. */
|
|
1871
|
+
scenarios: Array<{
|
|
1872
|
+
scenarioId: string;
|
|
1873
|
+
composite: number;
|
|
1874
|
+
notes?: string;
|
|
1875
|
+
}>;
|
|
1876
|
+
/** Proposer-supplied short label for the change. Present when the proposer
|
|
1877
|
+
* returned a `ProposedCandidate`; absent for bare-surface mutators. */
|
|
1878
|
+
label?: string;
|
|
1879
|
+
/** Proposer-supplied rationale — WHY this candidate was proposed. The
|
|
1880
|
+
* "because rationale Z" the audit requires to survive to the result.
|
|
1881
|
+
* Present when the proposer returned a `ProposedCandidate`. */
|
|
1882
|
+
rationale?: string;
|
|
1883
|
+
/** Exact structured cause threaded from the proposer, when available. */
|
|
1884
|
+
candidateRecord?: PolicyEditCandidateRecord;
|
|
1885
|
+
}
|
|
1886
|
+
interface CampaignAggregates {
|
|
1887
|
+
byJudge: Record<string, JudgeAggregate>;
|
|
1888
|
+
byScenario: Record<string, ScenarioAggregate>;
|
|
1889
|
+
/** Canonical campaign accounting, including worker and judge calls. */
|
|
1890
|
+
cost: CostLedgerSummary;
|
|
1891
|
+
/** Compatibility alias of `cost.totalCostUsd`. */
|
|
1892
|
+
totalCostUsd: number;
|
|
1893
|
+
cellsExecuted: number;
|
|
1894
|
+
cellsSkipped: number;
|
|
1895
|
+
cellsCached: number;
|
|
1896
|
+
cellsFailed: number;
|
|
1897
|
+
}
|
|
1898
|
+
interface CampaignResult<TArtifact = unknown, TScenario extends Scenario = Scenario> {
|
|
1899
|
+
/** sha256(scenarios, judges, dispatch source ref, optimizer config, seed). Stable identity for reruns. */
|
|
1900
|
+
manifestHash: string;
|
|
1901
|
+
seed: number;
|
|
1902
|
+
startedAt: string;
|
|
1903
|
+
endedAt: string;
|
|
1904
|
+
durationMs: number;
|
|
1905
|
+
cells: Array<CampaignCellResult<TArtifact>>;
|
|
1906
|
+
aggregates: CampaignAggregates;
|
|
1907
|
+
optimization?: {
|
|
1908
|
+
generations: GenerationRecord[];
|
|
1909
|
+
winnerSurfaceHash?: string;
|
|
1910
|
+
};
|
|
1911
|
+
gate?: GateResult;
|
|
1912
|
+
prUrl?: string;
|
|
1913
|
+
runDir: string;
|
|
1914
|
+
artifactsByPath: Record<string, string>;
|
|
1915
|
+
/** Substrate strips the input scenarios to id+kind for the result manifest;
|
|
1916
|
+
* consumers needing full payload look it up via the original input. The
|
|
1917
|
+
* type parameter `TScenario` is propagated for downstream consumers that
|
|
1918
|
+
* want narrowed types when extending `CampaignResult`. */
|
|
1919
|
+
scenarios: Array<Pick<TScenario, 'id' | 'kind'>>;
|
|
1920
|
+
}
|
|
1921
|
+
|
|
1922
|
+
/**
|
|
1923
|
+
* `CampaignStorage` — the filesystem seam `runCampaign` writes through
|
|
1924
|
+
* (run/cell dirs, the resumability cache, per-cell artifacts, trace spans).
|
|
1925
|
+
*
|
|
1926
|
+
* The default (`fsCampaignStorage`) is the Node filesystem — identical
|
|
1927
|
+
* behavior to the inline `node:fs` calls it replaces, so existing CLI
|
|
1928
|
+
* consumers are unaffected. `inMemoryCampaignStorage` keeps everything in a
|
|
1929
|
+
* `Map`, so the substrate runs in environments WITHOUT a filesystem
|
|
1930
|
+
* (Cloudflare Workers, Deno Deploy, other edge runtimes) — the campaign
|
|
1931
|
+
* still produces its `CampaignResult` (cells + aggregates) in memory;
|
|
1932
|
+
* artifacts/traces simply aren't persisted to disk.
|
|
1933
|
+
*
|
|
1934
|
+
* Paths are opaque keys to the in-memory adapter — it does not parse them,
|
|
1935
|
+
* so the same `join(...)`-built paths work unchanged across both adapters.
|
|
1936
|
+
*/
|
|
1937
|
+
interface CampaignStorage {
|
|
1938
|
+
/** Ensure a directory exists (recursive). No-op for in-memory. */
|
|
1939
|
+
ensureDir(dir: string): void;
|
|
1940
|
+
/** Does this path exist (as a written file or an ensured dir)? */
|
|
1941
|
+
exists(path: string): boolean;
|
|
1942
|
+
/** Read a UTF-8 file; `undefined` when missing or unreadable. */
|
|
1943
|
+
read(path: string): string | undefined;
|
|
1944
|
+
/** Write a file (string or bytes). Parent dir is assumed ensured. */
|
|
1945
|
+
write(path: string, content: string | Uint8Array): void;
|
|
1946
|
+
/** Append only when the current UTF-8 byte length matches `expectedBytes`.
|
|
1947
|
+
* Returns the new length, or undefined when another writer won. */
|
|
1948
|
+
append?(path: string, content: string, expectedBytes: number): number | undefined;
|
|
1949
|
+
}
|
|
1950
|
+
/** Node-filesystem storage — the default. Lazily requires `node:fs` so the
|
|
1951
|
+
* module imports cleanly in non-Node runtimes (where the caller passes
|
|
1952
|
+
* `inMemoryCampaignStorage` instead and never constructs this).
|
|
1953
|
+
*
|
|
1954
|
+
* `createRequire(import.meta.url)` is the ESM-native lazy require — a bare
|
|
1955
|
+
* `require` is a ReferenceError under `"type": "module"`, which is exactly
|
|
1956
|
+
* the shape this package publishes. */
|
|
1957
|
+
declare function fsCampaignStorage(): CampaignStorage;
|
|
1958
|
+
/** In-memory storage for filesystem-less runtimes. Artifacts + trace spans
|
|
1959
|
+
* live in a `Map` for the duration of the run; the `CampaignResult` is
|
|
1960
|
+
* fully populated, but nothing is persisted to disk. */
|
|
1961
|
+
declare function inMemoryCampaignStorage(): CampaignStorage;
|
|
1962
|
+
|
|
1963
|
+
/**
|
|
1964
|
+
* `runCampaign` — Pass A substrate primitive. ONE function that orchestrates
|
|
1965
|
+
* scenarios → dispatch → artifacts → judges → aggregates, with full
|
|
1966
|
+
* reproducibility (seed + manifest hash), cell-level resumability, bootstrap
|
|
1967
|
+
* CIs, and the `LabeledScenarioStore` capture flywheel.
|
|
1968
|
+
*
|
|
1969
|
+
* Improvement loops (optimizer / gate / autoOnPromote) ride on top of this
|
|
1970
|
+
* primitive but live in `presets/run-improvement-loop.ts`. This file keeps
|
|
1971
|
+
* the core orchestrator minimal — Phase 1 of the Pass A track.
|
|
1972
|
+
*/
|
|
1973
|
+
|
|
1974
|
+
interface RunCampaignOptions<TScenario extends Scenario, TArtifact> {
|
|
1975
|
+
scenarios: TScenario[];
|
|
1976
|
+
dispatch: DispatchFn<TScenario, TArtifact>;
|
|
1977
|
+
/**
|
|
1978
|
+
* Stable identity for the dispatch behavior, included in the manifest/cache
|
|
1979
|
+
* key. Set this when the same function name can run different models,
|
|
1980
|
+
* prompts, tools, or external config.
|
|
1981
|
+
*/
|
|
1982
|
+
dispatchRef?: string;
|
|
1983
|
+
judges?: JudgeConfig<TArtifact, TScenario>[];
|
|
1984
|
+
/** Required for reproducibility. Default 42. */
|
|
1985
|
+
seed?: number;
|
|
1986
|
+
/** Per-scenario replicates for CI bands. Default 1; raise to 5+ for
|
|
1987
|
+
* bootstrap-tight intervals on critical eval. */
|
|
1988
|
+
reps?: number;
|
|
1989
|
+
/** When true (default), completed cells are cached by
|
|
1990
|
+
* (manifestHash, scenarioId, rep, generation). Re-runs skip cached cells. */
|
|
1991
|
+
resumable?: boolean;
|
|
1992
|
+
/** Optional store — when present, every artifact + judge score is captured
|
|
1993
|
+
* with the configured `captureSource`. Capture is default ON; pass `'off'`
|
|
1994
|
+
* to disable. */
|
|
1995
|
+
labeledStore?: LabeledScenarioStore | 'off';
|
|
1996
|
+
captureSource?: 'production-trace' | 'eval-run' | 'manual' | 'red-team' | 'synthetic';
|
|
1997
|
+
captureSourceVersionHash?: string;
|
|
1998
|
+
/** Hard spend cap. Each paid call reserves its enforced maximum before dispatch. */
|
|
1999
|
+
costCeiling?: number;
|
|
2000
|
+
/** Shared spend account. Improvement loops pass one ledger through every
|
|
2001
|
+
* campaign so the ceiling and returned total are run-wide. */
|
|
2002
|
+
costLedger?: CostLedger;
|
|
2003
|
+
/** Attribution label for receipts recorded by this campaign. */
|
|
2004
|
+
costPhase?: string;
|
|
2005
|
+
/** Max concurrent cells. Default 2. */
|
|
2006
|
+
maxConcurrency?: number;
|
|
2007
|
+
/**
|
|
2008
|
+
* Per-cell dispatch deadline in ms. A `dispatch` that neither resolves nor
|
|
2009
|
+
* rejects within this window is a hang (a stalled model request, an
|
|
2010
|
+
* exhausted runtime resource, a backend that never closes its stream). When
|
|
2011
|
+
* set, the cell's `ctx.signal` is aborted and the cell is recorded as a LOUD
|
|
2012
|
+
* error (`dispatch exceeded <N>ms`) so the campaign proceeds and the failure
|
|
2013
|
+
* is visible — instead of one wedged cell silently hanging the whole run (and
|
|
2014
|
+
* every loop/CI job above it) forever. `undefined`/`0` = unbounded (legacy).
|
|
2015
|
+
*/
|
|
2016
|
+
dispatchTimeoutMs?: number;
|
|
2017
|
+
/** Required: where artifacts + traces land. A bare name (not an absolute path)
|
|
2018
|
+
* resolves to the shared `~/.tangle/traces/<repo>/runs/<name>` root so run
|
|
2019
|
+
* bundles never pollute a repo working tree. Pass an absolute path to override. */
|
|
2020
|
+
runDir: string;
|
|
2021
|
+
/** Subject repo for the shared run-dir root (defaults to the CWD basename).
|
|
2022
|
+
* Only consulted when `runDir` is a bare name. */
|
|
2023
|
+
repo?: string;
|
|
2024
|
+
/** Tracing posture. Default is the substrate's `FileSystemTraceStore` rooted
|
|
2025
|
+
* at `<runDir>/traces/`. `'off'` disables capture entirely — substrate
|
|
2026
|
+
* refuses this when the caller wires `autoOnPromote !== 'none'`. */
|
|
2027
|
+
tracing?: 'on' | 'off';
|
|
2028
|
+
/**
|
|
2029
|
+
* Per-cell usage expectation — the early, fine-grained sibling of the
|
|
2030
|
+
* batch `assertRealBackend` guard. A cell that produced an artifact (no
|
|
2031
|
+
* error) but reported `costUsd === 0` AND zero tokens is a stub: the
|
|
2032
|
+
* dispatch never reported LLM activity via `ctx.cost`. Modes:
|
|
2033
|
+
* - `'warn'` (default) — log the offending cell loudly, keep going.
|
|
2034
|
+
* - `'assert'` — throw `BackendIntegrityError` on the first such cell
|
|
2035
|
+
* (fail-fast; recommended for CI campaigns expecting real LLM calls).
|
|
2036
|
+
* - `'off'` — no check (replay / deterministic-only / offline analysis).
|
|
2037
|
+
*/
|
|
2038
|
+
expectUsage?: 'assert' | 'warn' | 'off';
|
|
2039
|
+
/** Test seam — override the wall clock for deterministic tests. */
|
|
2040
|
+
now?: () => Date;
|
|
2041
|
+
/** Test seam — override per-cell trace writer factory. */
|
|
2042
|
+
buildTraceWriter?: (cellId: string, dir: string) => CampaignTraceWriter;
|
|
2043
|
+
/** Storage backend for run/cell dirs, the resumability cache, artifacts,
|
|
2044
|
+
* and trace spans. Default: the Node filesystem (`fsCampaignStorage`).
|
|
2045
|
+
* Pass `inMemoryCampaignStorage()` to run in a filesystem-less runtime
|
|
2046
|
+
* (Cloudflare Workers, Deno, edge) — the `CampaignResult` is still
|
|
2047
|
+
* produced; artifacts/traces just aren't persisted to disk. */
|
|
2048
|
+
storage?: CampaignStorage;
|
|
2049
|
+
/**
|
|
2050
|
+
* Optional per-cell placement strategy. Returns an opaque string the
|
|
2051
|
+
* substrate forwards as `ctx.placement` to the Dispatch — placement-aware
|
|
2052
|
+
* Dispatches (e.g. `httpDispatch` from `/adapters/http`) use it to route
|
|
2053
|
+
* each cell to the right worker, region, or sandbox. When unset, every
|
|
2054
|
+
* cell receives `ctx.placement = undefined` and behaves identically to
|
|
2055
|
+
* the in-process case.
|
|
2056
|
+
*
|
|
2057
|
+
* @example
|
|
2058
|
+
* cellPlacement: ({ scenario }) => scenario.tags?.includes('eu') ? 'eu-west' : 'us-east'
|
|
2059
|
+
*/
|
|
2060
|
+
cellPlacement?: (input: {
|
|
2061
|
+
scenario: TScenario;
|
|
2062
|
+
rep: number;
|
|
2063
|
+
generation?: number;
|
|
2064
|
+
}) => string | undefined;
|
|
2065
|
+
}
|
|
2066
|
+
/**
|
|
2067
|
+
* Core campaign orchestrator: fan scenarios through dispatch, score with judges, aggregate bootstrap CIs, and persist reproducible `CampaignResult` records.
|
|
2068
|
+
*/
|
|
2069
|
+
declare function runCampaign<TScenario extends Scenario, TArtifact>(opts: RunCampaignOptions<TScenario, TArtifact>): Promise<CampaignResult<TArtifact, TScenario>>;
|
|
2070
|
+
|
|
2071
|
+
/**
|
|
2072
|
+
* `runEval` — the simplest preset over `runCampaign`. No optimizer, no
|
|
2073
|
+
* gate, no auto-PR. Just: run scenarios through dispatch, score with
|
|
2074
|
+
* judges, return CampaignResult.
|
|
2075
|
+
*
|
|
2076
|
+
* The 80% case for consumers who want a scorecard, not an improvement loop.
|
|
2077
|
+
*/
|
|
2078
|
+
|
|
2079
|
+
interface RunEvalOptions<TScenario extends Scenario, TArtifact> extends Omit<RunCampaignOptions<TScenario, TArtifact>, 'runDir'> {
|
|
2080
|
+
runDir: string;
|
|
2081
|
+
}
|
|
2082
|
+
/**
|
|
2083
|
+
* Simplest evaluation preset: run scenarios through dispatch, score with judges, and return a `CampaignResult` — no optimizer, no gate, no PR.
|
|
2084
|
+
*/
|
|
2085
|
+
declare function runEval<TScenario extends Scenario, TArtifact>(opts: RunEvalOptions<TScenario, TArtifact>): Promise<CampaignResult<TArtifact, TScenario>>;
|
|
2086
|
+
|
|
2087
|
+
/**
|
|
2088
|
+
* `openAutoPr` — thin shell-out helper for the `runImprovementLoop` preset's
|
|
2089
|
+
* `autoOnPromote: 'pr'` mode. Substitutes for the per-product PR-opening
|
|
2090
|
+
* code consumers duplicated 4 times. The PR body includes the campaign's
|
|
2091
|
+
* manifest hash, gate verdict, and scorecard summary so reviewers can see
|
|
2092
|
+
* exactly what was promoted + why.
|
|
2093
|
+
*
|
|
2094
|
+
* NOT a deploy mechanism — this only OPENS a PR. The human reviews + merges.
|
|
2095
|
+
* The Shape B (`autoOnPromote: 'config'`) live-runtime-mutation path is
|
|
2096
|
+
* deferred to Pass B with the full shadow / canary / rollback stack.
|
|
2097
|
+
*/
|
|
2098
|
+
|
|
2099
|
+
interface OpenAutoPrOptions<TArtifact, TScenario extends Scenario> {
|
|
2100
|
+
/** Campaign result to attach to the PR. */
|
|
2101
|
+
result: CampaignResult<TArtifact, TScenario>;
|
|
2102
|
+
/** Gate verdict explaining the promotion. Substrate refuses to open a PR
|
|
2103
|
+
* when `gate.decision !== 'ship'` — fails loud. */
|
|
2104
|
+
gate: GateResult;
|
|
2105
|
+
/** Promoted surface diff — typically the new system prompt addendum or
|
|
2106
|
+
* full profile diff. Substrate writes it as the PR body. */
|
|
2107
|
+
promotedDiff: string;
|
|
2108
|
+
/** GH owner/repo target (e.g., `tangle-network/gtm-agent`). */
|
|
2109
|
+
ghOwner: string;
|
|
2110
|
+
ghRepo: string;
|
|
2111
|
+
/** Branch name for the PR. Default `auto/<manifestHash[:12]>`. */
|
|
2112
|
+
branch?: string;
|
|
2113
|
+
/** PR title. Default includes manifest hash. */
|
|
2114
|
+
title?: string;
|
|
2115
|
+
/** Whether to actually open the PR or just dry-run. Default reads
|
|
2116
|
+
* `GH_AUTO_PR_TOKEN` env — present = open, absent = dry-run. */
|
|
2117
|
+
dryRun?: boolean;
|
|
2118
|
+
/** Test seam — substitute `gh pr create` invocation. */
|
|
2119
|
+
ghExec?: (args: string[]) => {
|
|
2120
|
+
stdout: string;
|
|
2121
|
+
stderr: string;
|
|
2122
|
+
status: number;
|
|
2123
|
+
};
|
|
2124
|
+
}
|
|
2125
|
+
interface OpenAutoPrResult {
|
|
2126
|
+
opened: boolean;
|
|
2127
|
+
prUrl?: string;
|
|
2128
|
+
dryRun: boolean;
|
|
2129
|
+
reason: string;
|
|
2130
|
+
}
|
|
2131
|
+
/**
|
|
2132
|
+
* Open a GitHub PR for a gate-approved surface promotion, attaching the manifest hash, gate verdict, and diff as the PR body.
|
|
2133
|
+
*/
|
|
2134
|
+
declare function openAutoPr<TArtifact, TScenario extends Scenario>(options: OpenAutoPrOptions<TArtifact, TScenario>): OpenAutoPrResult;
|
|
2135
|
+
|
|
2136
|
+
/**
|
|
2137
|
+
* `runOptimization` — the improvement loop body. Runs N generations: the
|
|
2138
|
+
* `SurfaceProposer` proposes K candidate surfaces per generation, each
|
|
2139
|
+
* candidate runs a campaign (the measurement), and only a candidate that beats
|
|
2140
|
+
* the single global incumbent becomes the next generation's parent.
|
|
2141
|
+
* Proposer-agnostic — the same loop runs an evolutionary population mutator
|
|
2142
|
+
* (`evolutionaryProposer`) or any reflective / agentic proposer; they differ
|
|
2143
|
+
* only in how `propose()` picks candidates.
|
|
2144
|
+
*
|
|
2145
|
+
* This is `runLoop`'s shape (plan → measure → decide) specialized to surface
|
|
2146
|
+
* improvement: `proposer.propose` = plan, `runCampaign` = the measurement
|
|
2147
|
+
* (which runs the worker behind `dispatch`), the mean-composite ranking = the
|
|
2148
|
+
* validator, `proposer.decide` = the stop check.
|
|
2149
|
+
*
|
|
2150
|
+
* The gated-promotion shell (`runImprovementLoop`) wraps this with a holdout
|
|
2151
|
+
* re-score + release gate + optional PR.
|
|
2152
|
+
*/
|
|
2153
|
+
|
|
2154
|
+
interface RunOptimizationBaseOptions<TScenario extends Scenario, TArtifact> extends Omit<RunCampaignOptions<TScenario, TArtifact>, 'dispatch'> {
|
|
2155
|
+
/** Initial mutable surface (typically system prompt or addendum). */
|
|
2156
|
+
baselineSurface: MutableSurface;
|
|
2157
|
+
/** Dispatcher that takes the CURRENT surface + scenario → artifact. */
|
|
2158
|
+
dispatchWithSurface: (surface: MutableSurface, scenario: TScenario, ctx: Parameters<RunCampaignOptions<TScenario, TArtifact>['dispatch']>[1]) => Promise<TArtifact>;
|
|
2159
|
+
/** The candidate-generation strategy. Wrap a population `Mutator` via
|
|
2160
|
+
* `evolutionaryProposer({ mutator })`, or pass any reflective / agentic
|
|
2161
|
+
* proposer that implements `SurfaceProposer`. */
|
|
2162
|
+
proposer: SurfaceProposer;
|
|
2163
|
+
populationSize: number;
|
|
2164
|
+
maxGenerations: number;
|
|
2165
|
+
/** @deprecated The loop has one global incumbent and can promote only the
|
|
2166
|
+
* single candidate that beats it. Retained for source compatibility. */
|
|
2167
|
+
promoteTopK?: number;
|
|
2168
|
+
/** DEPTH knob forwarded to the proposer's `propose()` — max iterations the
|
|
2169
|
+
* agentic generator may take per candidate. */
|
|
2170
|
+
maxImprovementShots?: number;
|
|
2171
|
+
/** Optional analysis report forwarded to `propose()`. Opaque here; the
|
|
2172
|
+
* proposer types it. */
|
|
2173
|
+
report?: unknown;
|
|
2174
|
+
/** Structured findings forwarded to `propose()` as `ctx.findings`. A
|
|
2175
|
+
* findings producer emits these from the
|
|
2176
|
+
* generation's traces; findings-grounded proposers consume them. Opaque here;
|
|
2177
|
+
* the proposer types its `TFindings`. Empty when no producer is wired. */
|
|
2178
|
+
findings?: unknown[];
|
|
2179
|
+
/** Per-generation findings producer. Runs once on the BASELINE campaign
|
|
2180
|
+
* (as `generation: -1`, the baseline convention) before generation 0
|
|
2181
|
+
* proposes — so even a single-generation run proposes with trace context —
|
|
2182
|
+
* and then after each generation's candidates are scored with that
|
|
2183
|
+
* generation's results; whatever it returns REPLACES `ctx.findings` for the
|
|
2184
|
+
* NEXT `propose()`, so the diagnosis is refreshed each round instead
|
|
2185
|
+
* of being a static one-shot. Generic by design: the substrate does not
|
|
2186
|
+
* import an analyst — the consumer plugs its trace-analyst registry / HALO
|
|
2187
|
+
* here (reading the per-candidate `runDir` traces). When absent, findings
|
|
2188
|
+
* stay the static `opts.findings`. */
|
|
2189
|
+
analyzeGeneration?: (input: {
|
|
2190
|
+
generation: number;
|
|
2191
|
+
runDir: string;
|
|
2192
|
+
candidates: Array<{
|
|
2193
|
+
surfaceHash: string;
|
|
2194
|
+
campaign: CampaignResult<TArtifact, TScenario>;
|
|
2195
|
+
composite: number;
|
|
2196
|
+
}>;
|
|
2197
|
+
history: GenerationRecord[];
|
|
2198
|
+
/** Shared run spend account and receipt attribution phase. */
|
|
2199
|
+
costLedger?: CostLedger;
|
|
2200
|
+
costPhase?: string;
|
|
2201
|
+
}) => Promise<unknown[]>;
|
|
2202
|
+
}
|
|
2203
|
+
type RunOptimizationOptions<TScenario extends Scenario, TArtifact> = RunOptimizationBaseOptions<TScenario, TArtifact>;
|
|
2204
|
+
interface RunOptimizationResult<TArtifact, TScenario extends Scenario> {
|
|
2205
|
+
generations: Array<{
|
|
2206
|
+
record: GenerationRecord;
|
|
2207
|
+
surfaces: Array<{
|
|
2208
|
+
surfaceHash: string;
|
|
2209
|
+
surface: MutableSurface;
|
|
2210
|
+
campaign: CampaignResult<TArtifact, TScenario>;
|
|
2211
|
+
}>;
|
|
2212
|
+
}>;
|
|
2213
|
+
winnerSurface: MutableSurface;
|
|
2214
|
+
winnerSurfaceHash: string;
|
|
2215
|
+
/** Proposer label for the promoted surface. Present when the winning
|
|
2216
|
+
* candidate came from a `ProposedCandidate` (a reflective proposer);
|
|
2217
|
+
* absent when the winner is the baseline or a bare-surface mutator. */
|
|
2218
|
+
winnerLabel?: string;
|
|
2219
|
+
/** Proposer rationale for the promoted surface — the "because Z" that
|
|
2220
|
+
* motivated the winning change. Survives to `SelfImproveResult` and the
|
|
2221
|
+
* emitted provenance record. Absent when the winner is the baseline. */
|
|
2222
|
+
winnerRationale?: string;
|
|
2223
|
+
baselineCampaign: CampaignResult<TArtifact, TScenario>;
|
|
2224
|
+
/** Run-wide spend, including agents, proposers, analysts, and judges. */
|
|
2225
|
+
cost: CostLedgerSummary;
|
|
2226
|
+
/** The GEPA Pareto frontier across every scored surface (baseline + all
|
|
2227
|
+
* generations) by per-scenario objective vector — the non-dominated set.
|
|
2228
|
+
* Each generation's `propose()` received the frontier-so-far as
|
|
2229
|
+
* `ctx.paretoParents`; this is the final frontier. A surface here that is
|
|
2230
|
+
* NOT the winner is uniquely best on some scenario the winner loses on. */
|
|
2231
|
+
paretoFrontier: ParetoParent[];
|
|
2232
|
+
}
|
|
2233
|
+
|
|
2234
|
+
/**
|
|
2235
|
+
* `runImprovementLoop` — the gated-promotion shell around the improvement
|
|
2236
|
+
* loop body (`runOptimization`). Proposes candidate surfaces via the
|
|
2237
|
+
* `SurfaceProposer`, re-scores the winner against the baseline on a
|
|
2238
|
+
* holdout set, runs the release gate, and optionally opens a PR.
|
|
2239
|
+
*
|
|
2240
|
+
* Role vocabulary (see docs/design/loop-taxonomy.md):
|
|
2241
|
+
* - PROPOSER = the `SurfaceProposer` (evolutionary GEPA mutator OR
|
|
2242
|
+
* reflective analyst). Proposes candidate SURFACES — the
|
|
2243
|
+
* worker's system prompt / tool config — NOT conversation
|
|
2244
|
+
* turns.
|
|
2245
|
+
* - MEASUREMENT= `runCampaign`. Scores one surface by running the worker
|
|
2246
|
+
* (via `dispatch`) over scenarios and judging the output.
|
|
2247
|
+
* - WORKER = the agent harness in the sandbox, invoked behind the
|
|
2248
|
+
* topology-opaque `dispatch` seam — never referenced here.
|
|
2249
|
+
*
|
|
2250
|
+
* Distinct from `runLoop` in `@tangle-network/agent-runtime`, which is the
|
|
2251
|
+
* INNER conversation loop (execution driver ↔ workers in a sandbox). `runImprovementLoop`
|
|
2252
|
+
* is the OUTER loop: it improves the surface that those workers run.
|
|
2253
|
+
*
|
|
2254
|
+
* Hard-refuses unsafe configurations:
|
|
2255
|
+
* - `tracing: 'off'` when a proposer is wired (improvement is unattributable)
|
|
2256
|
+
* - `autoOnPromote: 'config'` — DEFERRED to Pass B; v0.40 only ships
|
|
2257
|
+
* `'pr'` and `'none'`.
|
|
2258
|
+
*/
|
|
2259
|
+
|
|
2260
|
+
type RunImprovementLoopOptions<TScenario extends Scenario, TArtifact> = RunOptimizationOptions<TScenario, TArtifact> & {
|
|
2261
|
+
/** Holdout scenarios kept OUT of the training optimization pool — used
|
|
2262
|
+
* ONLY to score baseline vs winner for the gate. */
|
|
2263
|
+
holdoutScenarios: TScenario[];
|
|
2264
|
+
/** Promotion gate. Substrate strongly recommends `defaultProductionGate`
|
|
2265
|
+
* for production wiring (composes red-team / reward-hacking / canary /
|
|
2266
|
+
* heldout). */
|
|
2267
|
+
gate: Gate<TArtifact, TScenario>;
|
|
2268
|
+
/** What to do when the gate ships:
|
|
2269
|
+
* - `'pr'`: open a PR via `openAutoPr`
|
|
2270
|
+
* - `'none'`: just report — caller decides what to do with the winner
|
|
2271
|
+
* v0.40 does NOT support `'config'` (live-runtime self-mutation) —
|
|
2272
|
+
* deferred to Pass B behind safety stack. */
|
|
2273
|
+
autoOnPromote: 'pr' | 'none';
|
|
2274
|
+
/** GH owner / repo for the auto-PR. Required when autoOnPromote === 'pr'. */
|
|
2275
|
+
ghOwner?: string;
|
|
2276
|
+
ghRepo?: string;
|
|
2277
|
+
/** Optional render override — substrate writes a diff-shaped surface; pass
|
|
2278
|
+
* a function to format the promoted surface differently. */
|
|
2279
|
+
renderPromotedDiff?: (winnerSurface: MutableSurface, baselineSurface: MutableSurface) => string;
|
|
2280
|
+
/** Placebo control. When supplied AND the winner differs from baseline, the
|
|
2281
|
+
* loop scores a THIRD holdout arm: the winner surface with its content
|
|
2282
|
+
* footprint-matched-blanked by this function (typically via `neutralizeText`).
|
|
2283
|
+
* Its scores are exposed to the gate as `ctx.neutralizedJudgeScores`, letting
|
|
2284
|
+
* a `neutralizationGate` reject a win whose lift survives blanking the content
|
|
2285
|
+
* (decorative — driven by footprint, not content). Costs one extra holdout
|
|
2286
|
+
* campaign; omit to skip. Return a byte/layout-matched blank of the winner. */
|
|
2287
|
+
neutralize?: (winnerSurface: MutableSurface, baselineSurface: MutableSurface) => MutableSurface;
|
|
2288
|
+
};
|
|
2289
|
+
interface RunImprovementLoopResult<TArtifact, TScenario extends Scenario> extends RunOptimizationResult<TArtifact, TScenario> {
|
|
2290
|
+
baselineOnHoldout: CampaignResult<TArtifact, TScenario>;
|
|
2291
|
+
winnerOnHoldout: CampaignResult<TArtifact, TScenario>;
|
|
2292
|
+
gateResult: Awaited<ReturnType<Gate<TArtifact, TScenario>['decide']>>;
|
|
2293
|
+
/** Unified baseline→winner surface diff. Computed UNCONDITIONALLY (not only
|
|
2294
|
+
* when `autoOnPromote === 'pr'`) so the diff that the gate decided on is
|
|
2295
|
+
* always present on the result + in the emitted provenance record. Empty
|
|
2296
|
+
* string when winner == baseline (no change to diff). */
|
|
2297
|
+
promotedDiff: string;
|
|
2298
|
+
prResult?: ReturnType<typeof openAutoPr>;
|
|
2299
|
+
}
|
|
2300
|
+
/**
|
|
2301
|
+
* Gated-promotion shell over `runOptimization`: scores the winner against the baseline on a holdout set, runs the release gate, and optionally opens a PR.
|
|
2302
|
+
*/
|
|
2303
|
+
declare function runImprovementLoop<TScenario extends Scenario, TArtifact>(opts: RunImprovementLoopOptions<TScenario, TArtifact>): Promise<RunImprovementLoopResult<TArtifact, TScenario>>;
|
|
2304
|
+
|
|
2305
|
+
declare const REFERENCE_EQUIVALENCE_JUDGE_VERSION = "reference-equivalence-judge-v1-2026-07-13";
|
|
2306
|
+
declare const REFERENCE_EQUIVALENCE_INPUT_LIMITS: {
|
|
2307
|
+
readonly userRequest: 8000;
|
|
2308
|
+
readonly expectedAnswer: 32000;
|
|
2309
|
+
readonly candidateOutput: 32000;
|
|
2310
|
+
};
|
|
2311
|
+
interface ReferenceEquivalenceScenario extends Scenario {
|
|
2312
|
+
userRequest: string;
|
|
2313
|
+
expectedAnswer: string;
|
|
2314
|
+
}
|
|
2315
|
+
interface ReferenceEquivalenceJudgeInput {
|
|
2316
|
+
userRequest: string;
|
|
2317
|
+
expectedAnswer: string;
|
|
2318
|
+
candidateOutput: string;
|
|
2319
|
+
}
|
|
2320
|
+
interface ReferenceEquivalenceJudgeOptions {
|
|
2321
|
+
/** Injected transport. No implicit provider or credentials are selected. */
|
|
2322
|
+
chat: ChatClient;
|
|
2323
|
+
/** Falls back to the ChatClient's default model. */
|
|
2324
|
+
model?: string;
|
|
2325
|
+
/** Used only by the direct-call adapter. */
|
|
2326
|
+
signal?: AbortSignal;
|
|
2327
|
+
/** Optional receipt destination for direct calls; campaigns supply their own. */
|
|
2328
|
+
costLedger?: CostLedger;
|
|
2329
|
+
}
|
|
2330
|
+
interface ReferenceEquivalenceJudgeResult extends LlmCallMetadata {
|
|
2331
|
+
kind: 'reference-equivalence';
|
|
2332
|
+
version: string;
|
|
2333
|
+
score: number;
|
|
2334
|
+
rationale: string;
|
|
2335
|
+
}
|
|
2336
|
+
/** Build the campaign-native expected-answer judge. */
|
|
2337
|
+
declare function createReferenceEquivalenceJudge(options: ReferenceEquivalenceJudgeOptions): JudgeConfig<string, ReferenceEquivalenceScenario>;
|
|
2338
|
+
/** Direct-call adapter over the campaign judge for product callers. */
|
|
2339
|
+
declare function runReferenceEquivalenceJudge(input: ReferenceEquivalenceJudgeInput, options: ReferenceEquivalenceJudgeOptions): Promise<ReferenceEquivalenceJudgeResult>;
|
|
2340
|
+
|
|
2341
|
+
/**
|
|
2342
|
+
* `evolutionaryProposer` — adapts a stateless `Mutator` (population mutation:
|
|
2343
|
+
* GEPA / AxGEPA / reflective-mutation) into a `SurfaceProposer`. This is
|
|
2344
|
+
* the evolutionary strategy: each generation, mutate the current best surface
|
|
2345
|
+
* into N candidates, measure, select. No generation memory beyond the current
|
|
2346
|
+
* surface; the loop body handles ranking + promotion.
|
|
2347
|
+
*
|
|
2348
|
+
* The reflective alternative is agent-runtime's runtime proposer with a
|
|
2349
|
+
* `reflectiveGenerator` / `agenticGenerator`: it reasons over the report +
|
|
2350
|
+
* trace findings to propose targeted edits rather than blind mutations. Both
|
|
2351
|
+
* conform to `SurfaceProposer`; the improvement loop is identical either way.
|
|
2352
|
+
*/
|
|
2353
|
+
|
|
2354
|
+
interface EvolutionaryProposerOptions<TFindings = unknown> {
|
|
2355
|
+
mutator: Mutator<TFindings>;
|
|
2356
|
+
/** External findings fed to the mutator each generation. Default: []. */
|
|
2357
|
+
findings?: TFindings[];
|
|
2358
|
+
}
|
|
2359
|
+
/**
|
|
2360
|
+
* Wrap a stateless `Mutator` (GEPA, AxGEPA, reflective-mutation) as a `SurfaceProposer` that mutates the current best surface into N candidates each generation.
|
|
2361
|
+
*/
|
|
2362
|
+
declare function evolutionaryProposer<TFindings = unknown>(opts: EvolutionaryProposerOptions<TFindings>): SurfaceProposer<TFindings>;
|
|
2363
|
+
|
|
2364
|
+
/**
|
|
2365
|
+
* `gepaProposer` — a reflective `SurfaceProposer` for prompt-tier surfaces.
|
|
2366
|
+
* Each generation it reflects on the prior best candidate's per-scenario
|
|
2367
|
+
* scores + weakest dimensions, asks an LLM to propose targeted rewrites of
|
|
2368
|
+
* the current surface, and returns them as the next population.
|
|
2369
|
+
*
|
|
2370
|
+
* Maps onto the GEPA paper (Agrawal et al., arXiv:2507.19457):
|
|
2371
|
+
* - *Reflection*: each generation reflects on the best parent's weakest
|
|
2372
|
+
* dimensions + per-scenario top/bottom scores to propose targeted rewrites.
|
|
2373
|
+
* - *Pareto frontier*: `runOptimization` maintains the non-dominated set of
|
|
2374
|
+
* surfaces across generations (per-scenario objective vectors) and supplies
|
|
2375
|
+
* it as `ctx.paretoParents`. A surface uniquely best on one hard scenario
|
|
2376
|
+
* survives even when its mean composite is lower.
|
|
2377
|
+
* - *Combine complementary lessons*: when the frontier has >1 member, the
|
|
2378
|
+
* first population slot is a merge of those parents' strengths (one LLM
|
|
2379
|
+
* call citing each parent's winning scenarios). Toggle via `combineParents`.
|
|
2380
|
+
* Dominance is computed by the package-canonical `paretoFrontier` (`pareto.ts`).
|
|
2381
|
+
*
|
|
2382
|
+
* Optional `constraints` move structured-doc guards into the proposer
|
|
2383
|
+
* (preserve H2 section headings, cap sentence-level edits) — useful when
|
|
2384
|
+
* the surface IS a structured procedure like a SKILL.md / runbook /
|
|
2385
|
+
* judge rubric. When `constraints` is omitted, behavior is unchanged.
|
|
2386
|
+
*
|
|
2387
|
+
* The proposer is surface-agnostic — any string surface in any consumer opts
|
|
2388
|
+
* in by selecting it. Reuses the generic reflection primitive
|
|
2389
|
+
* (`buildReflectionPrompt` / `parseReflectionResponse`) and the router client.
|
|
2390
|
+
*
|
|
2391
|
+
* Earns its keep where there is real per-instance signal (which the
|
|
2392
|
+
* dimensional + per-scenario evidence + the `LabeledScenarioStore` flywheel
|
|
2393
|
+
* now provide). For thin-signal surfaces it degrades to plain reflection.
|
|
2394
|
+
* On generation 0 (no history) it reflects on the current surface against
|
|
2395
|
+
* the mutation primitives alone.
|
|
2396
|
+
*/
|
|
2397
|
+
|
|
2398
|
+
interface GepaProposerConstraints {
|
|
2399
|
+
/** H2 section headings that MUST appear unchanged in every candidate.
|
|
2400
|
+
* When set, the proposer auto-detects current H2s if this is empty AND
|
|
2401
|
+
* rejects any candidate that drops or renames a preserved heading.
|
|
2402
|
+
* Use when the surface is a structured doc (SKILL.md, runbook,
|
|
2403
|
+
* sectioned system prompt, judge rubric). */
|
|
2404
|
+
preserveSections?: string[];
|
|
2405
|
+
/** Maximum sentence-level edits per candidate vs the parent surface.
|
|
2406
|
+
* Rejection threshold = maxSentenceEdits × 2 (counts adds + removes).
|
|
2407
|
+
* Inspired by SkillOpt's edit-budget as a "textual learning rate."
|
|
2408
|
+
* Cap prevents an LLM rewrite from overwriting useful prior rules. */
|
|
2409
|
+
maxSentenceEdits?: number;
|
|
2410
|
+
}
|
|
2411
|
+
interface GepaProposerOptions {
|
|
2412
|
+
/** Router transport (apiKey/baseUrl). */
|
|
2413
|
+
llm: LlmClientOptions;
|
|
2414
|
+
/** Model that performs the reflection. */
|
|
2415
|
+
model: string;
|
|
2416
|
+
/** Optional ledger for direct proposer use. Campaign context takes precedence. */
|
|
2417
|
+
costLedger?: CostLedger;
|
|
2418
|
+
/** What is being optimized — appears in the reflection prompt for orientation. */
|
|
2419
|
+
target: string;
|
|
2420
|
+
/** Surface-specific mutation levers offered to the model. */
|
|
2421
|
+
mutationPrimitives?: string[];
|
|
2422
|
+
/** Top/bottom scenarios surfaced as evidence each generation. Default 3. */
|
|
2423
|
+
evidenceK?: number;
|
|
2424
|
+
/** Reflection sampling temperature. Default 0.7. */
|
|
2425
|
+
temperature?: number;
|
|
2426
|
+
/** Reflection max tokens. Default 6000. */
|
|
2427
|
+
maxTokens?: number;
|
|
2428
|
+
/** Structured-doc constraints. Candidates violating any are rejected
|
|
2429
|
+
* post-parse and dropped from the returned population. */
|
|
2430
|
+
constraints?: GepaProposerConstraints;
|
|
2431
|
+
/** GEPA combine-complementary-lessons: when the loop supplies a Pareto
|
|
2432
|
+
* frontier of >1 non-dominated parents (`ctx.paretoParents`), spend one
|
|
2433
|
+
* slot of the population on a merge of their strengths. Default `true` —
|
|
2434
|
+
* this is the GEPA-faithful behavior; the merge only fires once the
|
|
2435
|
+
* frontier has more than one member (generation ≥ 1). Set `false` for
|
|
2436
|
+
* pure single-parent reflection. */
|
|
2437
|
+
combineParents?: boolean;
|
|
2438
|
+
/** Cap on how many frontier parents feed one combine prompt (highest
|
|
2439
|
+
* composite first), to bound prompt size. Default 4. */
|
|
2440
|
+
combineMaxParents?: number;
|
|
2441
|
+
}
|
|
2442
|
+
/**
|
|
2443
|
+
* GEPA reflective proposer: each generation reflects on the weakest scenarios and dimensions to produce targeted prompt rewrites, optionally combining Pareto-frontier parents.
|
|
2444
|
+
*/
|
|
2445
|
+
declare function gepaProposer(opts: GepaProposerOptions): SurfaceProposer;
|
|
2446
|
+
|
|
2447
|
+
/**
|
|
2448
|
+
* Compose multiple `Gate` implementations — every gate must pass for the
|
|
2449
|
+
* composite to ship. Closes the alignment reviewer's "default-only
|
|
2450
|
+
* heldOutGate + costGate would happily promote a reward-hacked prompt"
|
|
2451
|
+
* concern by making safety gates first-class composable defaults.
|
|
2452
|
+
*/
|
|
2453
|
+
|
|
2454
|
+
/** Compose gates — all must `ship` for the composite to `ship`. First
|
|
2455
|
+
* non-ship verdict short-circuits the composite verdict, but ALL gates run
|
|
2456
|
+
* (so the result records every gate's reason — useful for diagnostics). */
|
|
2457
|
+
declare function composeGate<TArtifact = unknown, TScenario extends Scenario = Scenario>(...gates: Array<Gate<TArtifact, TScenario>>): Gate<TArtifact, TScenario>;
|
|
2458
|
+
|
|
2459
|
+
/**
|
|
2460
|
+
* Dataset — versioned, sliceable, content-hashed scenario collection.
|
|
2461
|
+
*
|
|
2462
|
+
* Scenarios stop being ephemeral arrays and become first-class
|
|
2463
|
+
* artifacts. Every Dataset carries:
|
|
2464
|
+
* - content hash (sha256 over canonicalized scenario array)
|
|
2465
|
+
* - provenance (contributor, createdAt, sourceUrl)
|
|
2466
|
+
* - split labels (train | dev | test | holdout)
|
|
2467
|
+
* - difficulty tiers (easy | medium | hard | extreme)
|
|
2468
|
+
* - tags (free-form, per-scenario)
|
|
2469
|
+
*
|
|
2470
|
+
* `Dataset.slice({ difficulty, split, holdout, seed })` returns a
|
|
2471
|
+
* deterministic, reproducible subset. Holdout slices are locked: you
|
|
2472
|
+
* can read them but `mutate` throws, which prevents "oh I'll just
|
|
2473
|
+
* tweak that one scenario" contamination drift.
|
|
2474
|
+
*/
|
|
2475
|
+
type DatasetSplit = 'train' | 'dev' | 'test' | 'holdout';
|
|
2476
|
+
type DatasetDifficulty = 'easy' | 'medium' | 'hard' | 'extreme';
|
|
2477
|
+
interface DatasetScenario {
|
|
2478
|
+
id: string;
|
|
2479
|
+
/** Arbitrary payload; the framework doesn't interpret it. */
|
|
2480
|
+
payload: unknown;
|
|
2481
|
+
split?: DatasetSplit;
|
|
2482
|
+
difficulty?: DatasetDifficulty;
|
|
2483
|
+
/** Canary token that MUST NOT round-trip through a correct agent output. */
|
|
2484
|
+
canary?: string;
|
|
2485
|
+
/**
|
|
2486
|
+
* Behavioral-canary forbidden pattern. A string OR a serialized regex
|
|
2487
|
+
* (`/.../flags`) that the agent under test MUST NOT emit. Used by
|
|
2488
|
+
* {@link import('./canary').checkBehavioralCanary | checkBehavioralCanary},
|
|
2489
|
+
* which inverts the contamination-style semantic: presence in the
|
|
2490
|
+
* agent output is a LEAK / failure, not a positive signal.
|
|
2491
|
+
*
|
|
2492
|
+
* Falls back to {@link canary} when omitted.
|
|
2493
|
+
*/
|
|
2494
|
+
forbiddenPattern?: string;
|
|
2495
|
+
tags?: Record<string, string>;
|
|
2496
|
+
}
|
|
2497
|
+
|
|
2498
|
+
/**
|
|
2499
|
+
* Red-team battery — adversarial scenario corpus with per-category
|
|
2500
|
+
* scorers.
|
|
2501
|
+
*
|
|
2502
|
+
* Every scenario is a `DatasetScenario` with a `redTeam: {...}` payload
|
|
2503
|
+
* that declares: category, expected refusal, attack surface. Scorers
|
|
2504
|
+
* inspect the agent's output + trajectory and return a PASS/FAIL per
|
|
2505
|
+
* category. Aggregated via `redTeamReport`.
|
|
2506
|
+
*
|
|
2507
|
+
* The built-in corpus is intentionally conservative — 40 cases
|
|
2508
|
+
* covering the most-seen attack patterns. Extend via `extendCorpus`.
|
|
2509
|
+
*/
|
|
2510
|
+
|
|
2511
|
+
type RedTeamCategory = 'prompt_injection_direct' | 'prompt_injection_indirect' | 'jailbreak_persona' | 'jailbreak_dan' | 'pii_leak' | 'permission_escalation' | 'data_exfil' | 'policy_override';
|
|
2512
|
+
interface RedTeamPayload {
|
|
2513
|
+
category: RedTeamCategory;
|
|
2514
|
+
/** The hostile input the agent sees. */
|
|
2515
|
+
input: string;
|
|
2516
|
+
/**
|
|
2517
|
+
* What a safe agent should do: "refuse" means explicit refusal,
|
|
2518
|
+
* "ignore" means process benignly without executing the attack.
|
|
2519
|
+
*/
|
|
2520
|
+
expected: 'refuse' | 'ignore';
|
|
2521
|
+
/** Substring(s) that MUST NOT appear in output. */
|
|
2522
|
+
forbiddenStrings?: string[];
|
|
2523
|
+
/** Tool names the agent MUST NOT call for this scenario. */
|
|
2524
|
+
forbiddenTools?: string[];
|
|
2525
|
+
}
|
|
2526
|
+
interface RedTeamCase extends DatasetScenario {
|
|
2527
|
+
payload: RedTeamPayload;
|
|
2528
|
+
}
|
|
2529
|
+
|
|
2530
|
+
/**
|
|
2531
|
+
* `defaultProductionGate` — composes the substrate's existing safety
|
|
2532
|
+
* primitives (red-team / reward-hacking / canary / heldout) into a single
|
|
2533
|
+
* Gate.decide shape. Closes the alignment + Anthropic-SI reviewers' "safety
|
|
2534
|
+
* primitives are off the critical path" blocker.
|
|
2535
|
+
*
|
|
2536
|
+
* The composition is opinionated — when consumers wire `runImprovementLoop`,
|
|
2537
|
+
* THIS gate is the default. Consumers can still pass a custom gate to
|
|
2538
|
+
* override; the recommended pattern is to compose THIS gate with whatever
|
|
2539
|
+
* extra domain-specific gates they need (`composeGate(defaultProductionGate(...), customGate)`).
|
|
2540
|
+
*/
|
|
2541
|
+
|
|
2542
|
+
interface DefaultProductionGateOptions {
|
|
2543
|
+
/** Required: scenarios held out from training; substrate compares
|
|
2544
|
+
* candidate-on-holdout vs baseline-on-holdout. */
|
|
2545
|
+
holdoutScenarios: Scenario[];
|
|
2546
|
+
/** Minimum held-out lift the **paired-bootstrap CI lower bound** must clear
|
|
2547
|
+
* to ship — NOT a point estimate. Default 0 ⇒ "confidently positive at the
|
|
2548
|
+
* confidence level". Interpreted in the judge's native composite scale (set
|
|
2549
|
+
* e.g. 2 for a 0-100 rubric to require a ≥2-point significant gain). */
|
|
2550
|
+
deltaThreshold?: number;
|
|
2551
|
+
/** Confidence level for the held-out + dimension bootstraps. Default 0.95. */
|
|
2552
|
+
confidence?: number;
|
|
2553
|
+
/** Bootstrap resamples. Default 2000. */
|
|
2554
|
+
bootstrapResamples?: number;
|
|
2555
|
+
/** Fixed bootstrap seed for a deterministic verdict. Default 1337. */
|
|
2556
|
+
bootstrapSeed?: number;
|
|
2557
|
+
/** Minimum paired holdout observations (scenarios × reps) before a
|
|
2558
|
+
* significance claim is allowed; below it the gate HOLDS with `few_runs`
|
|
2559
|
+
* rather than reading a degenerate CI. Default 3. */
|
|
2560
|
+
minProductiveRuns?: number;
|
|
2561
|
+
/** Ship statistic for the held-out significance test. Default `'mean'`
|
|
2562
|
+
* (tie-robust — see `heldoutSignificance`). Pass `'median'` for
|
|
2563
|
+
* outlier-robustness at the cost of tie-blindness. */
|
|
2564
|
+
heldoutStatistic?: 'mean' | 'median';
|
|
2565
|
+
/** Critical judge dimensions that must NOT significantly regress even when
|
|
2566
|
+
* the net composite rises (anti-Goodhart). The gate HOLDS if any listed
|
|
2567
|
+
* dimension's paired-delta CI lower bound < −`regressionTolerance`. E.g.
|
|
2568
|
+
* `['hallucination_free']` for a legal agent. */
|
|
2569
|
+
criticalDimensions?: string[];
|
|
2570
|
+
/** Tolerance for the per-dimension regression guard, in the dimension's
|
|
2571
|
+
* native scale. When omitted it auto-scales off observed magnitudes:
|
|
2572
|
+
* 0.05 on [0,1], 5 on 0-100. */
|
|
2573
|
+
regressionTolerance?: number;
|
|
2574
|
+
/** Total $ budget for ALL cells in this campaign — including baseline + candidate.
|
|
2575
|
+
* Composite verdict refuses to ship when spend exceeded budget. */
|
|
2576
|
+
budgetUsd?: number;
|
|
2577
|
+
/** Red-team cases to probe candidate outputs against. When omitted the
|
|
2578
|
+
* substrate uses `DEFAULT_RED_TEAM_CORPUS`. Provide a domain-specific
|
|
2579
|
+
* battery for tighter coverage. */
|
|
2580
|
+
redTeamBattery?: RedTeamCase[];
|
|
2581
|
+
/** Run records (oldest-first) needed for the reward-hacking detector.
|
|
2582
|
+
* Substrate populates from prior production-loop generations. */
|
|
2583
|
+
recentRuns?: RunRecord[];
|
|
2584
|
+
/** When true, the gate refuses to ship if the reward-hacking detector
|
|
2585
|
+
* fires at the `gaming` severity. Default true. */
|
|
2586
|
+
blockOnRewardHackingGaming?: boolean;
|
|
2587
|
+
}
|
|
2588
|
+
/**
|
|
2589
|
+
* Opinionated production gate composing held-out significance, red-team, reward-hacking, and canary checks into a single `Gate.decide` decision.
|
|
2590
|
+
*/
|
|
2591
|
+
declare function defaultProductionGate<TArtifact, TScenario extends Scenario>(options: DefaultProductionGateOptions): Gate<TArtifact, TScenario>;
|
|
2592
|
+
|
|
2593
|
+
/**
|
|
2594
|
+
* @module
|
|
2595
|
+
* Composable held-out promotion gate backed by paired bootstrap confidence.
|
|
2596
|
+
*
|
|
2597
|
+
* Pair by full `scenario:rep` cellId, bootstrap the paired candidate-minus-
|
|
2598
|
+
* baseline delta, and ship only when CI.low strictly clears the threshold with
|
|
2599
|
+
* at least `minProductiveRuns` paired observations.
|
|
2600
|
+
*
|
|
2601
|
+
* Use when you want held-out significance as ONE of N composed gates instead
|
|
2602
|
+
* of the full `defaultProductionGate` stack (which adds critical-dimension
|
|
2603
|
+
* regression + reward-hacking guards on top).
|
|
2604
|
+
*/
|
|
2605
|
+
|
|
2606
|
+
interface HeldOutGateOptions<TScenario extends Scenario = Scenario> {
|
|
2607
|
+
scenarios: TScenario[];
|
|
2608
|
+
/** Effect-size threshold the CI lower bound must clear, in the judge's native
|
|
2609
|
+
* scale. Default 0.5. Equality holds; CI.low must be greater than this value. */
|
|
2610
|
+
deltaThreshold?: number;
|
|
2611
|
+
/** Bootstrap CI confidence. Default 0.95. */
|
|
2612
|
+
confidence?: number;
|
|
2613
|
+
/** Minimum paired holdout observations to claim significance. Default 3. */
|
|
2614
|
+
minProductiveRuns?: number;
|
|
2615
|
+
/** Bootstrap resamples. Default 2000. */
|
|
2616
|
+
resamples?: number;
|
|
2617
|
+
/** Fixed bootstrap seed for deterministic verdicts. Default 1337. */
|
|
2618
|
+
bootstrapSeed?: number;
|
|
2619
|
+
}
|
|
2620
|
+
/**
|
|
2621
|
+
* Composable held-out gate: ships only when the PAIRED bootstrap CI lower bound
|
|
2622
|
+
* of the candidate-minus-baseline composite delta clears `deltaThreshold`.
|
|
2623
|
+
*/
|
|
2624
|
+
declare function heldOutGate<TArtifact, TScenario extends Scenario>(options: HeldOutGateOptions<TScenario>): Gate<TArtifact, TScenario>;
|
|
2625
|
+
|
|
2626
|
+
/**
|
|
2627
|
+
* Pareto frontier — multi-objective optimization over candidate runs.
|
|
2628
|
+
*
|
|
2629
|
+
* Lifted from ADC pareto.ts and blueprint-agent frontier.ts. When you're
|
|
2630
|
+
* trading off (cost, latency, quality) or (passRate, tokenBudget,
|
|
2631
|
+
* ttfb), you rarely have a single "winner" — you have a set of
|
|
2632
|
+
* non-dominated candidates. This module exposes:
|
|
2633
|
+
*
|
|
2634
|
+
* - `paretoFrontier`: filter a set of candidates to the non-dominated ones
|
|
2635
|
+
* - `dominates`: does A dominate B across all objectives?
|
|
2636
|
+
*
|
|
2637
|
+
* Each objective is declared with a direction: 'maximize' (higher=better)
|
|
2638
|
+
* or 'minimize' (lower=better). Candidates are any object; pass an
|
|
2639
|
+
* `objective(candidate)` accessor.
|
|
2640
|
+
*/
|
|
2641
|
+
type Direction = 'maximize' | 'minimize';
|
|
2642
|
+
|
|
2643
|
+
interface ContinuousAgreement {
|
|
2644
|
+
/** Cohen's κ_w with quadratic weights, computed on raw [0,1] scores. */
|
|
2645
|
+
weightedKappa: number;
|
|
2646
|
+
/** ICC(2,1): two-way random effects, absolute agreement, single rater. */
|
|
2647
|
+
icc: number;
|
|
2648
|
+
/** Pearson product-moment correlation (averaged over rater pairs if N>2). */
|
|
2649
|
+
pearson: number;
|
|
2650
|
+
/** Spearman rank correlation (averaged over rater pairs if N>2). */
|
|
2651
|
+
spearman: number;
|
|
2652
|
+
/** 95% bootstrap percentile CIs over items. */
|
|
2653
|
+
ci: {
|
|
2654
|
+
icc: [number, number];
|
|
2655
|
+
weightedKappa: [number, number];
|
|
2656
|
+
};
|
|
2657
|
+
/** Number of complete items (no NaN across raters). */
|
|
2658
|
+
n: number;
|
|
2659
|
+
/** Number of raters. */
|
|
2660
|
+
raters: number;
|
|
2661
|
+
}
|
|
2662
|
+
|
|
2663
|
+
interface PairedBootstrapResult {
|
|
2664
|
+
/** Number of paired observations. */
|
|
2665
|
+
n: number;
|
|
2666
|
+
/** Median of paired deltas (after − before). */
|
|
2667
|
+
median: number;
|
|
2668
|
+
/** Mean of paired deltas. */
|
|
2669
|
+
mean: number;
|
|
2670
|
+
/** Lower bound of the bootstrap CI on the chosen statistic. */
|
|
2671
|
+
low: number;
|
|
2672
|
+
/** Upper bound of the bootstrap CI on the chosen statistic. */
|
|
2673
|
+
high: number;
|
|
2674
|
+
/** Confidence level used (e.g. 0.95). */
|
|
2675
|
+
confidence: number;
|
|
2676
|
+
/** Number of bootstrap resamples used. */
|
|
2677
|
+
resamples: number;
|
|
2678
|
+
}
|
|
2679
|
+
|
|
2680
|
+
/**
|
|
2681
|
+
* Promotion policy over the evidence VECTOR — the substrate's answer to "never
|
|
2682
|
+
* collapse the multi-objective promotion decision into one scalar." A
|
|
2683
|
+
* `defaultProductionGate` is one opinionated composition; this module factors
|
|
2684
|
+
* the decision into two reusable pieces so MANY policies can compete over the
|
|
2685
|
+
* SAME evidence (the quant-desk pattern: one evidence bus, plural strategies):
|
|
2686
|
+
*
|
|
2687
|
+
* buildEvidenceVector(ctx, objectives, opts) -> EvidenceVector // the bus
|
|
2688
|
+
* PromotionPolicy = (ev: EvidenceVector) => GateResult // a strategy
|
|
2689
|
+
* paretoPolicy(ev) // the default strategy
|
|
2690
|
+
* paretoSignificanceGate(options): Gate // bus + policy as a Gate
|
|
2691
|
+
*
|
|
2692
|
+
* The Pareto policy is SYMMETRIC multi-objective: every objective is BOTH a
|
|
2693
|
+
* potential gain source AND a safety floor (unlike `defaultProductionGate`,
|
|
2694
|
+
* where only `composite` can win and `criticalDimensions` are pure floors). A
|
|
2695
|
+
* candidate ships iff it weakly DOMINATES the baseline at the confidence level —
|
|
2696
|
+
* no objective credibly worse (CI floor breach) AND at least one objective
|
|
2697
|
+
* credibly better (CI gain). Insufficient evidence on ANY axis -> need_more_work
|
|
2698
|
+
* (NOT folded into hold: "gather more reps" and "reject" are different actions).
|
|
2699
|
+
*
|
|
2700
|
+
* Cost/latency are NOT CI axes here — `GateContext` carries only an aggregate
|
|
2701
|
+
* per-side cost, no per-cell observation vector to bootstrap. Treat them as hard
|
|
2702
|
+
* constraints (compose with a budget gate via `composeGate`), not faked CIs.
|
|
2703
|
+
*/
|
|
2704
|
+
|
|
2705
|
+
/** Where an objective's per-cell scalar comes from. `composite` reads the
|
|
2706
|
+
* judge's composite; `dimension` reads a named per-dimension score. */
|
|
2707
|
+
type ObjectiveSource = {
|
|
2708
|
+
kind: 'composite';
|
|
2709
|
+
} | {
|
|
2710
|
+
kind: 'dimension';
|
|
2711
|
+
dimension: string;
|
|
2712
|
+
};
|
|
2713
|
+
interface PromotionObjective {
|
|
2714
|
+
/** Stable label used in reports + `contributingGates`. */
|
|
2715
|
+
name: string;
|
|
2716
|
+
source: ObjectiveSource;
|
|
2717
|
+
/** 'maximize' (quality dims) or 'minimize' (error/risk/length dims). Orients
|
|
2718
|
+
* the paired delta so a positive bootstrap always means "candidate better". */
|
|
2719
|
+
direction: Direction;
|
|
2720
|
+
/** The good-direction paired-delta CI lower bound must EXCEED this to count
|
|
2721
|
+
* as a significant gain on this axis. Interpreted in the judge's native
|
|
2722
|
+
* scale. Default 0 (⇒ "confidently better"). */
|
|
2723
|
+
gainThreshold?: number;
|
|
2724
|
+
/** A floor breach (regression) is declared when the good-direction CI lower
|
|
2725
|
+
* bound is below −floorTolerance. When omitted it auto-scales off observed
|
|
2726
|
+
* magnitudes (0.05 on [0,1], 5 on 0-100), matching `dimensionRegressions`. */
|
|
2727
|
+
floorTolerance?: number;
|
|
2728
|
+
}
|
|
2729
|
+
/** Per-axis verdict from the good-direction paired bootstrap. */
|
|
2730
|
+
type AxisVerdict = 'improved' | 'regressed' | 'flat' | 'few_runs';
|
|
2731
|
+
interface AxisEvidence {
|
|
2732
|
+
name: string;
|
|
2733
|
+
source: ObjectiveSource;
|
|
2734
|
+
direction: Direction;
|
|
2735
|
+
/** Paired bootstrap on the GOOD-DIRECTION delta (oriented by `direction`):
|
|
2736
|
+
* a positive value means the candidate is better on this axis. */
|
|
2737
|
+
bootstrap: PairedBootstrapResult;
|
|
2738
|
+
/** Paired observations contributing to this axis. */
|
|
2739
|
+
n: number;
|
|
2740
|
+
gainThreshold: number;
|
|
2741
|
+
floorTolerance: number;
|
|
2742
|
+
verdict: AxisVerdict;
|
|
2743
|
+
}
|
|
2744
|
+
interface EvidenceVector {
|
|
2745
|
+
/** One entry per objective — NOTHING averaged across axes. */
|
|
2746
|
+
axes: AxisEvidence[];
|
|
2747
|
+
/** Smallest paired n across axes that produced observations — the binding
|
|
2748
|
+
* evidence-sufficiency constraint. 0 when no axis produced observations. */
|
|
2749
|
+
minN: number;
|
|
2750
|
+
/** Aggregate per-side cost from the gate context (a constraint input, not a
|
|
2751
|
+
* CI axis — see the module header). */
|
|
2752
|
+
cost: {
|
|
2753
|
+
candidate: number;
|
|
2754
|
+
baseline: number;
|
|
2755
|
+
};
|
|
2756
|
+
}
|
|
2757
|
+
/** A promotion strategy: a pure function from the evidence vector to a verdict.
|
|
2758
|
+
* Many policies can run over the same `EvidenceVector` and disagree — that's
|
|
2759
|
+
* the point (competing strategies, shared evidence). */
|
|
2760
|
+
type PromotionPolicy = (ev: EvidenceVector) => GateResult;
|
|
2761
|
+
interface BuildEvidenceVectorOptions {
|
|
2762
|
+
/** Minimum paired observations before an axis can claim significance; below
|
|
2763
|
+
* it the axis is `few_runs`. Default 3. */
|
|
2764
|
+
minProductiveRuns?: number;
|
|
2765
|
+
/** Confidence level for every axis bootstrap. Default 0.95. */
|
|
2766
|
+
confidence?: number;
|
|
2767
|
+
/** Bootstrap resamples. Default 2000. */
|
|
2768
|
+
resamples?: number;
|
|
2769
|
+
/** Fixed bootstrap seed for a deterministic, reproducible verdict. Default 1337. */
|
|
2770
|
+
seed?: number;
|
|
2771
|
+
}
|
|
2772
|
+
/**
|
|
2773
|
+
* The Evidence Bus. For each objective, pair candidate vs baseline by full
|
|
2774
|
+
* cellId and bootstrap a CI on the good-direction paired delta. Reuses the
|
|
2775
|
+
* exact `pairHoldout` + `pairedBootstrap` machinery the held-out gate uses, so
|
|
2776
|
+
* a single source of truth governs pairing granularity + scale handling.
|
|
2777
|
+
*/
|
|
2778
|
+
declare function buildEvidenceVector<TArtifact, TScenario extends Scenario>(ctx: GateContext<TArtifact, TScenario>, objectives: PromotionObjective[], opts?: BuildEvidenceVectorOptions): EvidenceVector;
|
|
2779
|
+
/**
|
|
2780
|
+
* The default strategy: symmetric multi-objective Pareto significance. Ship iff
|
|
2781
|
+
* the candidate weakly dominates the baseline at the confidence level — no axis
|
|
2782
|
+
* credibly worse AND ≥1 axis credibly better. Floor breach on any axis → hold
|
|
2783
|
+
* (anti-Goodhart, dominates everything). Insufficient evidence on any axis →
|
|
2784
|
+
* need_more_work. Statistically equivalent → hold (never ship noise).
|
|
2785
|
+
*/
|
|
2786
|
+
declare const paretoPolicy: PromotionPolicy;
|
|
2787
|
+
interface ParetoSignificanceGateOptions extends BuildEvidenceVectorOptions {
|
|
2788
|
+
/** The objective vector. Every axis is both a gain source and a safety floor. */
|
|
2789
|
+
objectives: PromotionObjective[];
|
|
2790
|
+
/** Strategy applied to the evidence vector. Default `paretoPolicy`. Override
|
|
2791
|
+
* to run a stricter/looser strategy over the SAME bus (competing policies). */
|
|
2792
|
+
policy?: PromotionPolicy;
|
|
2793
|
+
/** Override the gate name in reports. */
|
|
2794
|
+
name?: string;
|
|
2795
|
+
}
|
|
2796
|
+
/**
|
|
2797
|
+
* Wrap the bus + a policy as a `Gate`. Plugs into the existing
|
|
2798
|
+
* `runImprovementLoop({ gate })` slot and composes via `composeGate`; default
|
|
2799
|
+
* loop behavior is unchanged because consumers opt in by passing this gate.
|
|
2800
|
+
*/
|
|
2801
|
+
declare function paretoSignificanceGate<TArtifact = unknown, TScenario extends Scenario = Scenario>(options: ParetoSignificanceGateOptions): Gate<TArtifact, TScenario>;
|
|
2802
|
+
|
|
2803
|
+
/**
|
|
2804
|
+
* OutcomeStore — deployment outcomes attached to Run IDs.
|
|
2805
|
+
*
|
|
2806
|
+
* Outcomes arrive asynchronously from production telemetry after the
|
|
2807
|
+
* eval run completed: user ratings, retention flags, conversion events,
|
|
2808
|
+
* revenue, support-ticket rate, anything a product team can measure.
|
|
2809
|
+
* The store is a peer to TraceStore — separate lifecycle, same runId
|
|
2810
|
+
* foreign key.
|
|
2811
|
+
*
|
|
2812
|
+
* The whole point of this module is to make the meta-eval correlation
|
|
2813
|
+
* question computable: `correlate(evalMetric, outcomeMetric) → r, ρ, n, CI`.
|
|
2814
|
+
*/
|
|
2815
|
+
interface DeploymentOutcome {
|
|
2816
|
+
runId: string;
|
|
2817
|
+
capturedAt: number;
|
|
2818
|
+
/** Numeric outcomes keyed by name — retention_7d, csat, revenue_usd, etc. */
|
|
2819
|
+
metrics: Record<string, number>;
|
|
2820
|
+
/** Dimensions for stratified analysis — cohort, region, user_segment. */
|
|
2821
|
+
labels?: Record<string, string>;
|
|
2822
|
+
/** Free-form provenance (source system, pipeline version). */
|
|
2823
|
+
source?: string;
|
|
2824
|
+
}
|
|
2825
|
+
interface OutcomeFilter {
|
|
2826
|
+
runIds?: string[];
|
|
2827
|
+
since?: number;
|
|
2828
|
+
until?: number;
|
|
2829
|
+
label?: {
|
|
2830
|
+
key: string;
|
|
2831
|
+
value: string;
|
|
2832
|
+
};
|
|
2833
|
+
source?: string;
|
|
2834
|
+
}
|
|
2835
|
+
interface OutcomeStore {
|
|
2836
|
+
append(outcome: DeploymentOutcome): Promise<void>;
|
|
2837
|
+
/** All outcomes attached to this run (a single run can have many — multiple
|
|
2838
|
+
* capture windows over deployment time). */
|
|
2839
|
+
forRun(runId: string): Promise<DeploymentOutcome[]>;
|
|
2840
|
+
list(filter?: OutcomeFilter): Promise<DeploymentOutcome[]>;
|
|
2841
|
+
}
|
|
2842
|
+
declare class InMemoryOutcomeStore implements OutcomeStore {
|
|
2843
|
+
private items;
|
|
2844
|
+
append(outcome: DeploymentOutcome): Promise<void>;
|
|
2845
|
+
forRun(runId: string): Promise<DeploymentOutcome[]>;
|
|
2846
|
+
list(filter?: OutcomeFilter): Promise<DeploymentOutcome[]>;
|
|
2847
|
+
}
|
|
2848
|
+
interface FileSystemOutcomeStoreOptions {
|
|
2849
|
+
dir: string;
|
|
2850
|
+
maxBytes?: number;
|
|
2851
|
+
}
|
|
2852
|
+
declare class FileSystemOutcomeStore implements OutcomeStore {
|
|
2853
|
+
private dir;
|
|
2854
|
+
private maxBytes;
|
|
2855
|
+
private memo?;
|
|
2856
|
+
private loaded;
|
|
2857
|
+
constructor(options: FileSystemOutcomeStoreOptions);
|
|
2858
|
+
private ensureDir;
|
|
2859
|
+
append(outcome: DeploymentOutcome): Promise<void>;
|
|
2860
|
+
private load;
|
|
2861
|
+
forRun(runId: string): Promise<DeploymentOutcome[]>;
|
|
2862
|
+
list(filter?: OutcomeFilter): Promise<DeploymentOutcome[]>;
|
|
2863
|
+
}
|
|
2864
|
+
|
|
2865
|
+
/**
|
|
2866
|
+
* Reporting helpers — production summaries and paper-quality figures — sit alongside `reporter.ts` rather
|
|
2867
|
+
* than replacing it.
|
|
2868
|
+
*
|
|
2869
|
+
* Three artefacts:
|
|
2870
|
+
*
|
|
2871
|
+
* - `summaryTable` Markdown table of per-candidate means,
|
|
2872
|
+
* 95% bootstrap CIs, BH-adjusted Wilcoxon
|
|
2873
|
+
* p-values, and Cohen's d versus a
|
|
2874
|
+
* comparator candidate.
|
|
2875
|
+
* - `paretoChart` Abstract spec for a cost vs quality
|
|
2876
|
+
* scatter, with gate decisions overlaid.
|
|
2877
|
+
* Returns numbers + labels — caller
|
|
2878
|
+
* chooses the plotting library.
|
|
2879
|
+
* - `gainHistogram`
|
|
2880
|
+
* Per-item paired holdout deltas as a
|
|
2881
|
+
* histogram spec (bins + counts + median +
|
|
2882
|
+
* CI). Same "data, not images" contract.
|
|
2883
|
+
*
|
|
2884
|
+
* The figure types are PlotSpecs — JSON-friendly, library-agnostic.
|
|
2885
|
+
* They aren't React components and they aren't PNGs; they are
|
|
2886
|
+
* what you'd hand to vega-lite, plotly, matplotlib, or your own
|
|
2887
|
+
* Canvas renderer to draw the actual figure.
|
|
2888
|
+
*/
|
|
2889
|
+
|
|
2890
|
+
interface ParetoPoint {
|
|
2891
|
+
candidateId: string;
|
|
2892
|
+
/** Mean USD cost per run on the chosen split. */
|
|
2893
|
+
cost: number;
|
|
2894
|
+
/** Mean score on the chosen split. */
|
|
2895
|
+
quality: number;
|
|
2896
|
+
/** Number of runs that informed this point. */
|
|
2897
|
+
n: number;
|
|
2898
|
+
/** Whether this candidate is on the Pareto frontier — high
|
|
2899
|
+
* quality, low cost, no dominator. */
|
|
2900
|
+
onFrontier: boolean;
|
|
2901
|
+
/** Optional gate verdict for this candidate, if a `GateDecision`
|
|
2902
|
+
* for it was passed in. */
|
|
2903
|
+
gate?: 'promote' | 'reject_few_runs' | 'reject_negative_delta' | 'reject_overfit_gap' | null;
|
|
2904
|
+
}
|
|
2905
|
+
interface ParetoFigureSpec {
|
|
2906
|
+
kind: 'pareto-cost-quality';
|
|
2907
|
+
split: 'search' | 'holdout';
|
|
2908
|
+
points: ParetoPoint[];
|
|
2909
|
+
axes: {
|
|
2910
|
+
x: 'costUsd';
|
|
2911
|
+
y: 'score';
|
|
2912
|
+
};
|
|
2913
|
+
}
|
|
2914
|
+
interface GainDistributionBin {
|
|
2915
|
+
/** Inclusive lower edge. */
|
|
2916
|
+
lo: number;
|
|
2917
|
+
/** Exclusive upper edge (or inclusive if it's the last bin). */
|
|
2918
|
+
hi: number;
|
|
2919
|
+
/** Number of pairs whose delta lands in this bin. */
|
|
2920
|
+
count: number;
|
|
2921
|
+
}
|
|
2922
|
+
|
|
2923
|
+
/**
|
|
2924
|
+
* # InsightReport — the rigorous decision packet for any set of agent runs.
|
|
2925
|
+
*
|
|
2926
|
+
* Returned by `analyzeRuns()` and embedded in `SelfImproveResult.insight` +
|
|
2927
|
+
* the hosted-tier `EvalRunEvent.insightReport`. One shape across two surfaces:
|
|
2928
|
+
*
|
|
2929
|
+
* - **Customer who has a closed loop** (`selfImprove`): the report ships
|
|
2930
|
+
* with the loop output. Their dashboard renders ship/hold + lift CI +
|
|
2931
|
+
* calibration + cluster + Pareto in one packet.
|
|
2932
|
+
* - **Customer who has observed runs but no loop** (`analyzeRuns` directly):
|
|
2933
|
+
* same packet from a `RunRecord[]` they already have — production traces,
|
|
2934
|
+
* approve/reject corpus, CSV gold set.
|
|
2935
|
+
*
|
|
2936
|
+
* Every field is optional except the distributional summary — fields are
|
|
2937
|
+
* populated when the input data supports them:
|
|
2938
|
+
*
|
|
2939
|
+
* - `lift` requires both baseline and candidate splits to be present.
|
|
2940
|
+
* - `interRater` requires multi-rater feedback (≥2 raters per run).
|
|
2941
|
+
* - `judges` populates per-judge stats only when the run records carry
|
|
2942
|
+
* `outcome.judgeScores`.
|
|
2943
|
+
* - `failureClusters` requires the optional `analystRegistry` to be wired.
|
|
2944
|
+
* - `contamination` requires canary scenarios to be passed in.
|
|
2945
|
+
* - `outcomeCorrelation` requires a downstream outcome signal.
|
|
2946
|
+
* - `sequential` requires the run set to be ordered (treats them as a
|
|
2947
|
+
* stream and emits an anytime-valid interim decision).
|
|
2948
|
+
*
|
|
2949
|
+
* Consumers read the `recommendations` array first — that's the
|
|
2950
|
+
* actionable layer, ranked by priority. The numeric sections back it up.
|
|
2951
|
+
*/
|
|
2952
|
+
|
|
2953
|
+
interface InsightReport {
|
|
2954
|
+
/** Number of runs analyzed. */
|
|
2955
|
+
n: number;
|
|
2956
|
+
/** Runtime facts carried by the run records. These describe execution,
|
|
2957
|
+
* not task quality: duration, queueing, token categories, models, and
|
|
2958
|
+
* explicitly recorded failures. */
|
|
2959
|
+
execution: ExecutionInsight;
|
|
2960
|
+
/** Composite-score distribution across all runs. Always present. */
|
|
2961
|
+
composite: ScalarDistribution;
|
|
2962
|
+
/** Per-dimension distributions for every dimension that appeared in any
|
|
2963
|
+
* run's judge scores. Empty when no judge scores were recorded. */
|
|
2964
|
+
perDimension: Record<string, ScalarDistribution>;
|
|
2965
|
+
/** Cost/quality distribution and Pareto frontier. */
|
|
2966
|
+
costQuality: {
|
|
2967
|
+
cost: ScalarDistribution;
|
|
2968
|
+
pareto: ParetoFigureSpec;
|
|
2969
|
+
/** Cost source coverage. `uncaptured` rows are excluded from the USD
|
|
2970
|
+
* distribution and Pareto chart; observed and estimated totals remain
|
|
2971
|
+
* separate so reports never present estimates as billed spend. */
|
|
2972
|
+
provenance?: CostProvenanceSummary;
|
|
2973
|
+
/** Set when the cost/quality view is degraded because the input data
|
|
2974
|
+
* doesn't fully support it — e.g. all `costUsd` were zero, or only a
|
|
2975
|
+
* single candidate appears (so the Pareto is a single point). The
|
|
2976
|
+
* named fields name the degraded sub-view, free-text the reason. */
|
|
2977
|
+
degraded?: {
|
|
2978
|
+
cost?: string;
|
|
2979
|
+
pareto?: string;
|
|
2980
|
+
};
|
|
2981
|
+
};
|
|
2982
|
+
/** Per-judge calibration + bias detection. Populated for every judge name
|
|
2983
|
+
* that appears in `outcome.judgeScores`. Bias fields require either a
|
|
2984
|
+
* gold reference or multi-rater data. */
|
|
2985
|
+
judges: Record<string, JudgeInsight>;
|
|
2986
|
+
/** Inter-rater agreement when multiple judges scored the same runs.
|
|
2987
|
+
* Includes pairwise kappa and the specific run ids where raters
|
|
2988
|
+
* disagree — the cases worth a human meeting. */
|
|
2989
|
+
interRater?: InterRaterInsight;
|
|
2990
|
+
/** Pairwise lift (baseline → candidate) with bootstrap CI. Present when
|
|
2991
|
+
* `RunRecord.splitTag` includes both `holdout` and search/dev splits,
|
|
2992
|
+
* or when caller passes an explicit baseline/candidate split. */
|
|
2993
|
+
lift?: LiftInsight;
|
|
2994
|
+
/** Failure clusters with exemplars. Populated when an AnalystRegistry
|
|
2995
|
+
* is wired in `analyzeRuns({ analyst })`. */
|
|
2996
|
+
failureClusters?: FailureClusterInsight;
|
|
2997
|
+
/** Canary leak count + holdout audit status. Populated when canary
|
|
2998
|
+
* scenarios are passed in. */
|
|
2999
|
+
contamination?: ContaminationInsight;
|
|
3000
|
+
/** Correlation between judge composite and a downstream outcome the
|
|
3001
|
+
* caller supplies (engagement, revenue, downstream pass rate, etc.).
|
|
3002
|
+
* When present, the optional reward model is the model that maps
|
|
3003
|
+
* judge scores → predicted outcome. */
|
|
3004
|
+
outcomeCorrelation?: OutcomeCorrelationInsight;
|
|
3005
|
+
/** Aggregate release-readiness summary. A consumer needing the full
|
|
3006
|
+
* substrate `ReleaseConfidenceScorecard` (SLO-axis evaluation,
|
|
3007
|
+
* ActionableSideInfo bag) calls `evaluateReleaseConfidence()` directly;
|
|
3008
|
+
* this summary captures the analyzeRuns-derived axes. */
|
|
3009
|
+
release: ReleaseSummary;
|
|
3010
|
+
/** Delta vs a prior period when `baselineRuns` is passed. Per-metric
|
|
3011
|
+
* current vs baseline with Welch CI + Cohen's d + significance flag.
|
|
3012
|
+
* Answers "did my last change help?" — the customer-conversion question.
|
|
3013
|
+
* Surfaced metrics: composite, cost, duration, tokenUsage, plus any
|
|
3014
|
+
* per-dimension judge metric present in both windows. */
|
|
3015
|
+
priorPeriodComparison?: PriorPeriodComparison;
|
|
3016
|
+
/** Model-free failure-mode breakdown from `RunRecord.failureMode`, ranked
|
|
3017
|
+
* by count descending. Present when any run carries a `failureMode`.
|
|
3018
|
+
* Complements `failureClusters` (LLM-semantic) with the structured tags
|
|
3019
|
+
* the harness already recorded — actionable with no analyst wired. */
|
|
3020
|
+
failureModes?: FailureModeTally[];
|
|
3021
|
+
/** Top-N actionable recommendations, ranked by priority. The packet's
|
|
3022
|
+
* human-readable layer; the numeric sections are the evidence. */
|
|
3023
|
+
recommendations: Recommendation[];
|
|
3024
|
+
}
|
|
3025
|
+
interface CostProvenanceSummary {
|
|
3026
|
+
observed: {
|
|
3027
|
+
n: number;
|
|
3028
|
+
totalUsd: number;
|
|
3029
|
+
};
|
|
3030
|
+
estimated: {
|
|
3031
|
+
n: number;
|
|
3032
|
+
totalUsd: number;
|
|
3033
|
+
};
|
|
3034
|
+
uncaptured: {
|
|
3035
|
+
n: number;
|
|
3036
|
+
};
|
|
3037
|
+
knownFraction: number;
|
|
3038
|
+
}
|
|
3039
|
+
interface ExecutionInsight {
|
|
3040
|
+
/** End-to-end wall time for every run. */
|
|
3041
|
+
durationMs: ScalarDistribution;
|
|
3042
|
+
/** Queue time for the subset of runs that recorded it. */
|
|
3043
|
+
queueMs: ScalarDistribution;
|
|
3044
|
+
/** Token distributions plus corpus totals. Optional token categories use
|
|
3045
|
+
* distribution `n` to disclose how many runs recorded that category. */
|
|
3046
|
+
tokenUsage: TokenUsageInsight;
|
|
3047
|
+
/** Usage reported only by orchestration or agent aggregate spans.
|
|
3048
|
+
* Kept separate because it may duplicate model-call telemetry in other traces. */
|
|
3049
|
+
aggregateUsage: {
|
|
3050
|
+
runs: number;
|
|
3051
|
+
tokenUsage: TokenUsageInsight;
|
|
3052
|
+
costUsd: ScalarDistribution;
|
|
3053
|
+
totalCostUsd: number;
|
|
3054
|
+
};
|
|
3055
|
+
/** Stable model counts, largest cohort first. */
|
|
3056
|
+
models: Array<{
|
|
3057
|
+
model: string;
|
|
3058
|
+
runs: number;
|
|
3059
|
+
}>;
|
|
3060
|
+
/** Model-call coverage. `events` is available only from producers that
|
|
3061
|
+
* record `outcome.raw.llm_span_count`; `runs` also recognizes non-zero
|
|
3062
|
+
* token usage from other producers. */
|
|
3063
|
+
modelCalls: {
|
|
3064
|
+
runs: number;
|
|
3065
|
+
events: number;
|
|
3066
|
+
reportingRuns: number;
|
|
3067
|
+
};
|
|
3068
|
+
/** Failure counts remain separate from outcome scores. `reportedErrorEvents`
|
|
3069
|
+
* sums `outcome.raw.error_span_count` only where a producer supplied it. */
|
|
3070
|
+
failures: {
|
|
3071
|
+
runs: number;
|
|
3072
|
+
fraction: number;
|
|
3073
|
+
reportedErrorEvents: number;
|
|
3074
|
+
reportingRuns: number;
|
|
3075
|
+
};
|
|
3076
|
+
}
|
|
3077
|
+
interface TokenUsageInsight {
|
|
3078
|
+
input: ScalarDistribution;
|
|
3079
|
+
output: ScalarDistribution;
|
|
3080
|
+
reasoning: ScalarDistribution;
|
|
3081
|
+
cached: ScalarDistribution;
|
|
3082
|
+
cacheWrite: ScalarDistribution;
|
|
3083
|
+
totals: {
|
|
3084
|
+
input: number;
|
|
3085
|
+
output: number;
|
|
3086
|
+
reasoning: number;
|
|
3087
|
+
cached: number;
|
|
3088
|
+
cacheWrite: number;
|
|
3089
|
+
};
|
|
3090
|
+
}
|
|
3091
|
+
/** Distributional summary of a scalar-valued metric. */
|
|
3092
|
+
interface ScalarDistribution {
|
|
3093
|
+
/** Sample count after dropping non-finite values. */
|
|
3094
|
+
n: number;
|
|
3095
|
+
mean: number;
|
|
3096
|
+
p50: number;
|
|
3097
|
+
p95: number;
|
|
3098
|
+
stddev: number;
|
|
3099
|
+
min: number;
|
|
3100
|
+
max: number;
|
|
3101
|
+
/** Histogram bins using `agent-eval`'s `gainHistogram` primitive. */
|
|
3102
|
+
histogram: GainDistributionBin[];
|
|
3103
|
+
/** Worst-N runs by score, ascending. Populated for the composite
|
|
3104
|
+
* distribution so the report names the runs a customer should
|
|
3105
|
+
* inspect first. Undefined when the distribution was computed from a
|
|
3106
|
+
* raw value list with no run identity (e.g. cost). */
|
|
3107
|
+
tailRuns?: Array<{
|
|
3108
|
+
runId: string;
|
|
3109
|
+
score: number;
|
|
3110
|
+
}>;
|
|
3111
|
+
}
|
|
3112
|
+
interface JudgeInsight {
|
|
3113
|
+
/** Number of times this judge scored a run. */
|
|
3114
|
+
n: number;
|
|
3115
|
+
/** Mean composite over this judge's runs. */
|
|
3116
|
+
meanScore: number;
|
|
3117
|
+
/** Calibration against a gold reference, when provided. Cohen's κ for
|
|
3118
|
+
* binary thresholding + continuous agreement metrics. */
|
|
3119
|
+
calibration?: ContinuousAgreement;
|
|
3120
|
+
/** Positional bias — when the judge sees options in different orders,
|
|
3121
|
+
* do its preferences track the content or the position? */
|
|
3122
|
+
positionalBias?: number;
|
|
3123
|
+
/** Self-preference — when the judge sees its own model's output vs a
|
|
3124
|
+
* competitor, does it over-pick its own? */
|
|
3125
|
+
selfPreference?: number;
|
|
3126
|
+
/** Verbosity bias — does the judge reward longer outputs regardless of
|
|
3127
|
+
* quality? */
|
|
3128
|
+
verbosityBias?: number;
|
|
3129
|
+
}
|
|
3130
|
+
interface InterRaterInsight {
|
|
3131
|
+
/** Number of raters whose scores were aggregated. */
|
|
3132
|
+
raters: number;
|
|
3133
|
+
/** Number of runs every rater scored. */
|
|
3134
|
+
jointlyRated: number;
|
|
3135
|
+
/** Cohen's κ averaged across rater pairs. */
|
|
3136
|
+
kappa: number;
|
|
3137
|
+
/** Pairwise κ per rater pair (key = `"raterA::raterB"`). */
|
|
3138
|
+
perPair: Record<string, number>;
|
|
3139
|
+
/** Run ids where raters disagree the most — the high-value triage list. */
|
|
3140
|
+
disagreementCases: Array<{
|
|
3141
|
+
runId: string;
|
|
3142
|
+
ratings: Array<{
|
|
3143
|
+
rater: string;
|
|
3144
|
+
score: number;
|
|
3145
|
+
}>;
|
|
3146
|
+
range: number;
|
|
3147
|
+
}>;
|
|
3148
|
+
}
|
|
3149
|
+
interface LiftInsight {
|
|
3150
|
+
baselineMean: number;
|
|
3151
|
+
candidateMean: number;
|
|
3152
|
+
/** Candidate − baseline. */
|
|
3153
|
+
delta: number;
|
|
3154
|
+
/** Lower / upper bound of bootstrap CI on the delta. */
|
|
3155
|
+
ci95: [number, number];
|
|
3156
|
+
/** Paired-t-test p-value. */
|
|
3157
|
+
pValue: number;
|
|
3158
|
+
/** Number of paired observations. */
|
|
3159
|
+
n: number;
|
|
3160
|
+
/** Cohen's d for the delta. */
|
|
3161
|
+
cohensD: number;
|
|
3162
|
+
/** Minimum detectable effect at current n, 80% power. */
|
|
3163
|
+
mde: number;
|
|
3164
|
+
/** Sample size needed to detect the observed delta at 80% power. */
|
|
3165
|
+
requiredN: number;
|
|
3166
|
+
}
|
|
3167
|
+
interface FailureClusterInsight {
|
|
3168
|
+
/** All clusters identified by the registry, ranked by share descending. */
|
|
3169
|
+
clusters: Array<{
|
|
3170
|
+
id: string;
|
|
3171
|
+
name: string;
|
|
3172
|
+
/** Fraction of failed runs in this cluster, 0..1. */
|
|
3173
|
+
share: number;
|
|
3174
|
+
/** Exemplar `runId`s (≤ 5) the consumer can drill into. */
|
|
3175
|
+
exemplars: string[];
|
|
3176
|
+
/** Short LLM-generated suggested fix when the registry supports it. */
|
|
3177
|
+
suggestedFix?: string;
|
|
3178
|
+
}>;
|
|
3179
|
+
totalFailures: number;
|
|
3180
|
+
}
|
|
3181
|
+
/** Model-free failure breakdown over the structured `RunRecord.failureMode`
|
|
3182
|
+
* enum. Unlike `failureClusters` (semantic, requires an LLM analyst), this
|
|
3183
|
+
* is computed directly from the tags the harness already recorded — so a
|
|
3184
|
+
* customer ingesting one batch with no judge/analyst still learns which
|
|
3185
|
+
* named failure dominates. */
|
|
3186
|
+
interface FailureModeTally {
|
|
3187
|
+
/** The `failureMode` tag. */
|
|
3188
|
+
mode: string;
|
|
3189
|
+
/** Number of runs carrying this tag. */
|
|
3190
|
+
count: number;
|
|
3191
|
+
/** Share of the whole corpus, 0..1. */
|
|
3192
|
+
share: number;
|
|
3193
|
+
}
|
|
3194
|
+
interface ContaminationInsight {
|
|
3195
|
+
/** Canary phrases that leaked into outputs. */
|
|
3196
|
+
leaks: number;
|
|
3197
|
+
/** Holdout audit verdict — did any holdout-tagged run end up in the
|
|
3198
|
+
* search/dev pool, or vice versa? */
|
|
3199
|
+
holdoutAuditPassed: boolean;
|
|
3200
|
+
details?: Array<{
|
|
3201
|
+
runId: string;
|
|
3202
|
+
canary: string;
|
|
3203
|
+
matched: string;
|
|
3204
|
+
}>;
|
|
3205
|
+
}
|
|
3206
|
+
interface OutcomeCorrelationInsight {
|
|
3207
|
+
/** What outcome the consumer is correlating against (e.g.
|
|
3208
|
+
* `'engagement_rate'`, `'approval_rate'`, `'downstream_pass'`). */
|
|
3209
|
+
metric: string;
|
|
3210
|
+
/** Number of (run, outcome) pairs used. */
|
|
3211
|
+
n: number;
|
|
3212
|
+
/** Pearson correlation between composite score and outcome. */
|
|
3213
|
+
pearson: number;
|
|
3214
|
+
/** Spearman rank correlation — robust to monotonic non-linearity. */
|
|
3215
|
+
spearman: number;
|
|
3216
|
+
/** When present, the simple linear reward model fit to the data. */
|
|
3217
|
+
rewardModel?: {
|
|
3218
|
+
intercept: number;
|
|
3219
|
+
slope: number;
|
|
3220
|
+
r2: number;
|
|
3221
|
+
};
|
|
3222
|
+
}
|
|
3223
|
+
interface ReleaseSummary {
|
|
3224
|
+
/** Overall verdict across axes — fail if any axis fails, else warn if any
|
|
3225
|
+
* warns, else pass. */
|
|
3226
|
+
status: 'pass' | 'warn' | 'fail';
|
|
3227
|
+
axes: Array<{
|
|
3228
|
+
name: 'quality-lift' | 'contamination' | 'composite-distribution';
|
|
3229
|
+
status: 'pass' | 'warn' | 'fail';
|
|
3230
|
+
detail: string;
|
|
3231
|
+
}>;
|
|
3232
|
+
/** Free-form issues surfaced beyond the standard axes. Empty by default;
|
|
3233
|
+
* consumers can post-process to populate. */
|
|
3234
|
+
issues: string[];
|
|
3235
|
+
}
|
|
3236
|
+
interface MetricDelta {
|
|
3237
|
+
/** Current-period mean. */
|
|
3238
|
+
current: number;
|
|
3239
|
+
/** Baseline-period mean. */
|
|
3240
|
+
baseline: number;
|
|
3241
|
+
/** current - baseline. Positive means improved (or, for cost/duration,
|
|
3242
|
+
* the consumer-side interpretation: "higher current" — semantic
|
|
3243
|
+
* direction depends on the metric). */
|
|
3244
|
+
delta: number;
|
|
3245
|
+
/** Welch 95% confidence interval on the delta. Two-sample, unpaired —
|
|
3246
|
+
* the baseline and current run sets may have different scenarios. */
|
|
3247
|
+
ci95: [number, number];
|
|
3248
|
+
/** Welch t-test p-value (two-sided). */
|
|
3249
|
+
pValue: number;
|
|
3250
|
+
/** Cohen's d (pooled stddev). Effect size, signed. */
|
|
3251
|
+
cohensD: number;
|
|
3252
|
+
/** Sample sizes. */
|
|
3253
|
+
baselineN: number;
|
|
3254
|
+
currentN: number;
|
|
3255
|
+
/** True when p < 0.05 AND |d| >= 0.2 (small-effect threshold). The
|
|
3256
|
+
* conjunction prevents large-effect-but-noisy and significant-but-
|
|
3257
|
+
* tiny from triggering recommendations. */
|
|
3258
|
+
significant: boolean;
|
|
3259
|
+
}
|
|
3260
|
+
interface PriorPeriodComparison {
|
|
3261
|
+
/** Sample counts. */
|
|
3262
|
+
baselineN: number;
|
|
3263
|
+
currentN: number;
|
|
3264
|
+
/** Optional human-readable label — "vs prior 7 days", "vs v3 release". */
|
|
3265
|
+
windowLabel?: string;
|
|
3266
|
+
/** Every metric we could compare. Keys: 'composite', 'cost', 'duration',
|
|
3267
|
+
* 'tokenUsage' for always-present ones; per-dimension keys when both
|
|
3268
|
+
* windows have judge scores on the same dimension. */
|
|
3269
|
+
metrics: Record<string, MetricDelta>;
|
|
3270
|
+
/** Metric names where current is significantly WORSE than baseline.
|
|
3271
|
+
* Direction-aware: for cost/duration, higher current = worse. */
|
|
3272
|
+
regressedMetrics: string[];
|
|
3273
|
+
/** Metric names where current is significantly BETTER than baseline. */
|
|
3274
|
+
improvedMetrics: string[];
|
|
3275
|
+
}
|
|
3276
|
+
interface Recommendation {
|
|
3277
|
+
priority: 'critical' | 'high' | 'medium' | 'low';
|
|
3278
|
+
kind: 'ship' | 'hold' | 'investigate' | 'fix' | 'recalibrate' | 'expand-corpus';
|
|
3279
|
+
title: string;
|
|
3280
|
+
detail: string;
|
|
3281
|
+
/** Optional pointer back into the report for the evidence. */
|
|
3282
|
+
evidencePath?: string;
|
|
3283
|
+
}
|
|
3284
|
+
|
|
3285
|
+
/**
|
|
3286
|
+
* # Hosted-tier wire format — the schema that EVERY orchestrator (ours,
|
|
3287
|
+
* a partner's self-hosted one, a future open implementation) must accept.
|
|
3288
|
+
*
|
|
3289
|
+
* **Stability:** every type in this file is committed under semver. New
|
|
3290
|
+
* minors only ADD optional fields. Breaking changes mean a major bump
|
|
3291
|
+
* (`HostedWireVersion` literal increment).
|
|
3292
|
+
*
|
|
3293
|
+
* The wire format is two event streams in one transport:
|
|
3294
|
+
*
|
|
3295
|
+
* 1. **Eval-run events** (`POST /v1/ingest/eval-runs`). Posted when a
|
|
3296
|
+
* campaign / improvement-loop completes (or per-generation if
|
|
3297
|
+
* streaming). Carries the structured result + per-cell scores +
|
|
3298
|
+
* surface diffs the orchestrator stores for the dashboard.
|
|
3299
|
+
*
|
|
3300
|
+
* 2. **Trace spans** (`POST /v1/ingest/traces`). Standard OTLP-shaped
|
|
3301
|
+
* spans with a few additional attributes so the orchestrator can
|
|
3302
|
+
* pivot from eval-run → underlying execution. Compatible with any
|
|
3303
|
+
* OTel collector.
|
|
3304
|
+
*
|
|
3305
|
+
* Both endpoints are authenticated with a bearer token + a tenant id
|
|
3306
|
+
* header. Tenants isolate everything downstream of ingest; no tenant
|
|
3307
|
+
* ever sees another tenant's data.
|
|
3308
|
+
*/
|
|
3309
|
+
|
|
3310
|
+
/** Lifecycle stages of an eval-run as the substrate reports them. */
|
|
3311
|
+
type EvalRunStatus = 'started' | 'baseline-complete' | 'generation-complete' | 'gate-decided' | 'finished' | 'errored';
|
|
3312
|
+
interface EvalRunCellScore {
|
|
3313
|
+
/** Stable scenario id from the consumer's scenario set. */
|
|
3314
|
+
scenarioId: string;
|
|
3315
|
+
/** Repetition index when reps > 1; 0 for the default. */
|
|
3316
|
+
rep: number;
|
|
3317
|
+
/** Composite score across all judges + dimensions for this cell. */
|
|
3318
|
+
compositeMean: number;
|
|
3319
|
+
/** Per-judge → per-dimension scores; null where the judge did not run. */
|
|
3320
|
+
dimensions: Record<string, Record<string, number>>;
|
|
3321
|
+
/** Per-cell error message if the dispatch threw. Null on success. */
|
|
3322
|
+
errorMessage?: string;
|
|
3323
|
+
}
|
|
3324
|
+
interface EvalRunGenerationSnapshot {
|
|
3325
|
+
/** Generation index. 0 is baseline. */
|
|
3326
|
+
index: number;
|
|
3327
|
+
/** Candidate surface fingerprint (stable hash) — pivot key into the
|
|
3328
|
+
* trace stream to fetch the underlying execution. */
|
|
3329
|
+
surfaceHash: string;
|
|
3330
|
+
/** The candidate surface itself. May be omitted to avoid PII when the
|
|
3331
|
+
* consumer prefers not to ship verbatim prompts. */
|
|
3332
|
+
surface?: MutableSurface;
|
|
3333
|
+
/** Per-cell scores for this generation. */
|
|
3334
|
+
cells: EvalRunCellScore[];
|
|
3335
|
+
/** Aggregate composite mean across all cells in this generation. */
|
|
3336
|
+
compositeMean: number;
|
|
3337
|
+
/** Total $ spent across this generation. */
|
|
3338
|
+
costUsd: number;
|
|
3339
|
+
/** Wall-clock duration of this generation. */
|
|
3340
|
+
durationMs: number;
|
|
3341
|
+
}
|
|
3342
|
+
/**
|
|
3343
|
+
* The top-level eval-run event. One ingest call per logical eval-run;
|
|
3344
|
+
* generations stream in incrementally via repeated calls with the same
|
|
3345
|
+
* `runId`. The orchestrator deduplicates by `(runId, generation.index)`.
|
|
3346
|
+
*/
|
|
3347
|
+
interface EvalRunEvent {
|
|
3348
|
+
/** Stable run id (the substrate's `runId`). UUID or substrate-generated. */
|
|
3349
|
+
runId: string;
|
|
3350
|
+
/** Where this run was happening — derived from `RunCampaignOptions.runDir`. */
|
|
3351
|
+
runDir: string;
|
|
3352
|
+
/** ISO-8601 timestamp the substrate recorded the event. */
|
|
3353
|
+
timestamp: string;
|
|
3354
|
+
/** Lifecycle stage this event represents. */
|
|
3355
|
+
status: EvalRunStatus;
|
|
3356
|
+
/** Free-form consumer tags (env, branch, model id, etc.). Searchable. */
|
|
3357
|
+
labels: Record<string, string>;
|
|
3358
|
+
/** Baseline campaign snapshot. Present when status >= baseline-complete. */
|
|
3359
|
+
baseline?: EvalRunGenerationSnapshot;
|
|
3360
|
+
/** Per-generation snapshots. Streams in; orchestrator appends. */
|
|
3361
|
+
generations: EvalRunGenerationSnapshot[];
|
|
3362
|
+
/** Final gate decision. Present when status >= gate-decided. */
|
|
3363
|
+
gateDecision?: GateDecision;
|
|
3364
|
+
/** Held-out lift = winner-on-holdout - baseline-on-holdout. */
|
|
3365
|
+
holdoutLift?: number;
|
|
3366
|
+
/** Total $ spent across baseline + every generation. */
|
|
3367
|
+
totalCostUsd: number;
|
|
3368
|
+
/** Total wall-clock duration. */
|
|
3369
|
+
totalDurationMs: number;
|
|
3370
|
+
/** Error message if status === 'errored'. */
|
|
3371
|
+
errorMessage?: string;
|
|
3372
|
+
/** Rigor packet emitted alongside the run — distributional summary,
|
|
3373
|
+
* paired-bootstrap lift CI, judge stats, inter-rater agreement,
|
|
3374
|
+
* contamination check, failure clusters (when an analyst is wired),
|
|
3375
|
+
* outcome correlation (when downstream signal is supplied), and the
|
|
3376
|
+
* recommendations the dashboard surfaces verbatim. Additive; older
|
|
3377
|
+
* clients that don't know about this field continue to work. */
|
|
3378
|
+
insightReport?: InsightReport;
|
|
3379
|
+
}
|
|
3380
|
+
/**
|
|
3381
|
+
* OTel-shape span with a few additional attributes for eval-run pivoting.
|
|
3382
|
+
* Compatible with any OTLP collector — `name`, `traceId`, `spanId`,
|
|
3383
|
+
* `startTimeUnixNano`, `endTimeUnixNano`, `attributes` are stock OTel.
|
|
3384
|
+
*/
|
|
3385
|
+
interface TraceSpanEvent {
|
|
3386
|
+
traceId: string;
|
|
3387
|
+
spanId: string;
|
|
3388
|
+
parentSpanId?: string;
|
|
3389
|
+
name: string;
|
|
3390
|
+
startTimeUnixNano: number;
|
|
3391
|
+
endTimeUnixNano: number;
|
|
3392
|
+
attributes: Record<string, string | number | boolean>;
|
|
3393
|
+
events?: Array<{
|
|
3394
|
+
timeUnixNano: number;
|
|
3395
|
+
name: string;
|
|
3396
|
+
attributes?: Record<string, string | number | boolean>;
|
|
3397
|
+
}>;
|
|
3398
|
+
status?: {
|
|
3399
|
+
code: 'OK' | 'ERROR' | 'UNSET';
|
|
3400
|
+
message?: string;
|
|
3401
|
+
};
|
|
3402
|
+
/** Pivot back into the eval-run stream. */
|
|
3403
|
+
'tangle.runId'?: string;
|
|
3404
|
+
/** Pivot to the specific generation. */
|
|
3405
|
+
'tangle.generation'?: number;
|
|
3406
|
+
/** Pivot to the specific cell. */
|
|
3407
|
+
'tangle.cellId'?: string;
|
|
3408
|
+
/** Pivot to the specific scenario. */
|
|
3409
|
+
'tangle.scenarioId'?: string;
|
|
3410
|
+
}
|
|
3411
|
+
|
|
3412
|
+
/**
|
|
3413
|
+
* # Hosted-tier ingest client.
|
|
3414
|
+
*
|
|
3415
|
+
* Ships eval-run events + trace spans to any orchestrator (ours, a
|
|
3416
|
+
* partner's self-hosted one, or a future open implementation) that
|
|
3417
|
+
* speaks the wire format in `./types.ts`.
|
|
3418
|
+
*
|
|
3419
|
+
* Three modes:
|
|
3420
|
+
* - **Ours:** point at `https://orchestrator.tangle.tools` (the host root —
|
|
3421
|
+
* the client appends the versioned `/v1/ingest/...` path itself; a trailing
|
|
3422
|
+
* `/v1` on the endpoint is tolerated and normalized away). We handle ingest
|
|
3423
|
+
* + storage + dashboard.
|
|
3424
|
+
* - **Self-hosted:** point at whatever URL runs the reference receiver
|
|
3425
|
+
* from `examples/hosted-ingest-server/`.
|
|
3426
|
+
* - **Off (default):** when `hostedTenant` is unset, nothing is sent.
|
|
3427
|
+
* Everything stays local.
|
|
3428
|
+
*/
|
|
3429
|
+
|
|
3430
|
+
interface HostedTenant {
|
|
3431
|
+
/** Orchestrator endpoint base URL (no trailing slash). Required. */
|
|
3432
|
+
endpoint: string;
|
|
3433
|
+
/** Bearer token issued by the orchestrator. Required. */
|
|
3434
|
+
apiKey: string;
|
|
3435
|
+
/** Tenant id — the orchestrator's primary key for this consumer. Required. */
|
|
3436
|
+
tenantId: string;
|
|
3437
|
+
/** Optional `fetch` override (auth wrappers, custom agent, test mocks). */
|
|
3438
|
+
fetchImpl?: typeof fetch;
|
|
3439
|
+
/** Per-call timeout in ms. Default 30s. */
|
|
3440
|
+
timeoutMs?: number;
|
|
3441
|
+
/** Retries on 5xx / network errors. Default 2. */
|
|
3442
|
+
retries?: number;
|
|
3443
|
+
}
|
|
3444
|
+
|
|
3445
|
+
interface PowerPreflight {
|
|
3446
|
+
/** Paired observations the comparison will have. */
|
|
3447
|
+
n: number;
|
|
3448
|
+
/** Baseline per-cell composite standard deviation (the variance the effect must beat). */
|
|
3449
|
+
sd: number;
|
|
3450
|
+
/** Minimum detectable lift: the smallest TRUE effect the gate could ship at this budget. */
|
|
3451
|
+
mde: number;
|
|
3452
|
+
/** Baseline holdout composite mean. */
|
|
3453
|
+
baselineMean: number;
|
|
3454
|
+
/** Headroom to a perfect 1.0 composite (the largest achievable lift on a [0,1] judge). */
|
|
3455
|
+
headroom: number;
|
|
3456
|
+
/** True when even the largest achievable effect (headroom) is below the MDE —
|
|
3457
|
+
* the run is structurally unable to ship regardless of proposal quality.
|
|
3458
|
+
* Only asserted for [0,1]-scaled judges (see `scaleAssumed`). */
|
|
3459
|
+
underpowered: boolean;
|
|
3460
|
+
/** True when composites look [0,1]-scaled; headroom/underpowered are only
|
|
3461
|
+
* meaningful under that convention (0-100 judges get mde/sd/n but no verdict). */
|
|
3462
|
+
scaleAssumed: boolean;
|
|
3463
|
+
deltaThreshold: number;
|
|
3464
|
+
confidence: number;
|
|
3465
|
+
/** Set when the holdout shares the gate's scoring channel: more cells cannot
|
|
3466
|
+
* buy back systematic judge bias — treat the MDE as a lower bound. */
|
|
3467
|
+
sharedChannelCaveat?: string;
|
|
3468
|
+
/** One actionable sentence for humans and logs. */
|
|
3469
|
+
recommendation: string;
|
|
3470
|
+
}
|
|
3471
|
+
|
|
3472
|
+
/**
|
|
3473
|
+
* Loop provenance — the durable, queryable record of WHAT a self-improvement
|
|
3474
|
+
* loop did and WHY, plus the OTel spans that let an OTLP collector pivot from
|
|
3475
|
+
* an eval-run to the underlying candidate→cell→gate→promote chain.
|
|
3476
|
+
*
|
|
3477
|
+
* Two artifacts, one source of truth:
|
|
3478
|
+
*
|
|
3479
|
+
* 1. `LoopProvenanceRecord` — a structured JSON record capturing every
|
|
3480
|
+
* candidate (surfaceHash + label + rationale + structured cause), its measured composite,
|
|
3481
|
+
* the gate decision + reasons + delta, the held-out lift, the explicit
|
|
3482
|
+
* baseline→candidate diff, and BACKEND PROVENANCE (the
|
|
3483
|
+
* `assertRealBackend` verdict + worker call count + model). This is the
|
|
3484
|
+
* ingestable audit artifact: the +lift recomputes from it, the "because
|
|
3485
|
+
* Z" rationale survives in it, and a stub backend is detectable from it.
|
|
3486
|
+
*
|
|
3487
|
+
* 2. `loopProvenanceSpans()` — the same chain emitted as OTLP-ingestable
|
|
3488
|
+
* `TraceSpanEvent`s, pivoted on the substrate's standard
|
|
3489
|
+
* `tangle.runId` / `tangle.scenarioId` / `tangle.cellId` /
|
|
3490
|
+
* `tangle.generation` attributes (the same pivots `/adapters/otel`
|
|
3491
|
+
* reads). The hosted `/v1/ingest/traces` endpoint receives the FULL loop,
|
|
3492
|
+
* not just the `cost.*` spans `runCampaign` already emits per cell.
|
|
3493
|
+
*
|
|
3494
|
+
* The record is built from the substrate's own loop result + the per-call
|
|
3495
|
+
* `RunRecord`s the worker emitted — no new measurement, no recomputation that
|
|
3496
|
+
* could drift from what the gate actually saw.
|
|
3497
|
+
*/
|
|
3498
|
+
|
|
3499
|
+
interface LoopProvenanceCandidate {
|
|
3500
|
+
/** Generation index this candidate was proposed in. */
|
|
3501
|
+
generation: number;
|
|
3502
|
+
/** 16-char loop-identity fingerprint (matches `GenerationCandidate.surfaceHash`). */
|
|
3503
|
+
surfaceHash: string;
|
|
3504
|
+
/** Full sha256 content hash — byte-identical-verifiable. */
|
|
3505
|
+
contentHash: string;
|
|
3506
|
+
/** Proposer label, when the proposer returned a `ProposedCandidate`. */
|
|
3507
|
+
label?: string;
|
|
3508
|
+
/** Proposer rationale — the "because Z". When the proposer returned a bare
|
|
3509
|
+
* surface (blind mutator) this is absent. */
|
|
3510
|
+
rationale?: string;
|
|
3511
|
+
/** Exact validated cause when the proposer emitted a structured record. */
|
|
3512
|
+
candidateRecord?: PolicyEditCandidateRecord;
|
|
3513
|
+
/** Exact complete incumbent this candidate mutated. */
|
|
3514
|
+
parentSurfaceHash: string;
|
|
3515
|
+
/** Search-split composite of the exact parent. */
|
|
3516
|
+
parentComposite: number;
|
|
3517
|
+
/** Search-split composite change relative to the exact parent. */
|
|
3518
|
+
observedDeltaFromParent?: number;
|
|
3519
|
+
/** Whether the candidate completed every designed cell and could be selected. */
|
|
3520
|
+
eligibleForPromotion: boolean;
|
|
3521
|
+
/** Designed-denominator receipt retained even for incomplete candidates. */
|
|
3522
|
+
coverage: NonNullable<GenerationCandidate['coverage']>;
|
|
3523
|
+
/** Mean composite this candidate scored on the search split. */
|
|
3524
|
+
composite: number;
|
|
3525
|
+
/** Whether this candidate was promoted out of its generation. */
|
|
3526
|
+
promoted: boolean;
|
|
3527
|
+
}
|
|
3528
|
+
interface LoopProvenanceBackend {
|
|
3529
|
+
/** `assertRealBackend`-grade verdict over the worker call records. */
|
|
3530
|
+
verdict: 'real' | 'mixed' | 'stub';
|
|
3531
|
+
/** Number of worker LLM calls captured (the audit's "worker call count"). */
|
|
3532
|
+
workerCallCount: number;
|
|
3533
|
+
/** Distinct model ids observed across worker calls. */
|
|
3534
|
+
models: string[];
|
|
3535
|
+
totalInputTokens: number;
|
|
3536
|
+
totalOutputTokens: number;
|
|
3537
|
+
totalCostUsd: number;
|
|
3538
|
+
}
|
|
3539
|
+
/**
|
|
3540
|
+
* The durable provenance record. Aligns to the hosted `EvalRunEvent` path but
|
|
3541
|
+
* ADDS the rationale + the explicit baseline→candidate diff (both omitted from
|
|
3542
|
+
* the bare hosted event) + backend provenance.
|
|
3543
|
+
*/
|
|
3544
|
+
interface LoopProvenanceRecord {
|
|
3545
|
+
schema: 'tangle.loop-provenance.v3';
|
|
3546
|
+
runId: string;
|
|
3547
|
+
runDir: string;
|
|
3548
|
+
timestamp: string;
|
|
3549
|
+
/** Baseline + winner surface content hashes — distinguishable, byte-verifiable. */
|
|
3550
|
+
baselineContentHash: string;
|
|
3551
|
+
winnerContentHash: string;
|
|
3552
|
+
/** Proposer label/rationale for the promoted change. Absent ⇒ winner == baseline. */
|
|
3553
|
+
winnerLabel?: string;
|
|
3554
|
+
winnerRationale?: string;
|
|
3555
|
+
/** The explicit baseline→winner unified diff the gate decided on. */
|
|
3556
|
+
diff: string;
|
|
3557
|
+
/** Every candidate across every generation, with its rationale and structured cause. */
|
|
3558
|
+
candidates: LoopProvenanceCandidate[];
|
|
3559
|
+
/** Baseline composite on the search split that generated the candidates. */
|
|
3560
|
+
baselineSearchComposite: number;
|
|
3561
|
+
/** The gate verdict — decision + reasons + contributing gates + delta. */
|
|
3562
|
+
gate: {
|
|
3563
|
+
decision: GateDecision;
|
|
3564
|
+
reasons: string[];
|
|
3565
|
+
delta?: number;
|
|
3566
|
+
contributingGates: Array<{
|
|
3567
|
+
name: string;
|
|
3568
|
+
passed: boolean;
|
|
3569
|
+
}>;
|
|
3570
|
+
};
|
|
3571
|
+
/** baseline-on-holdout composite mean. */
|
|
3572
|
+
baselineHoldoutComposite: number;
|
|
3573
|
+
/** winner-on-holdout composite mean. */
|
|
3574
|
+
winnerHoldoutComposite: number;
|
|
3575
|
+
/** winnerHoldout - baselineHoldout — RECOMPUTABLE from this record. */
|
|
3576
|
+
heldOutLift: number;
|
|
3577
|
+
/** Backend provenance: stub-vs-real verdict + worker call count + models. */
|
|
3578
|
+
backend: LoopProvenanceBackend;
|
|
3579
|
+
totalCostUsd: number;
|
|
3580
|
+
totalDurationMs: number;
|
|
3581
|
+
}
|
|
37
3582
|
|
|
38
3583
|
/**
|
|
39
3584
|
* # `selfImprove()` - the one-call improvement loop.
|
|
@@ -392,6 +3937,348 @@ interface DefinedAgentEval<TScenario extends Scenario, TArtifact> {
|
|
|
392
3937
|
*/
|
|
393
3938
|
declare function defineAgentEval<TScenario extends Scenario, TArtifact>(defaults: DefineAgentEvalOptions<TScenario, TArtifact>): DefinedAgentEval<TScenario, TArtifact>;
|
|
394
3939
|
|
|
3940
|
+
/**
|
|
3941
|
+
* Typed Ax output for analyst findings.
|
|
3942
|
+
*
|
|
3943
|
+
* Replaces the legacy `findings:string[]` pattern (where every bullet
|
|
3944
|
+
* became a flat-severity `AnalystFinding`) with a structured object
|
|
3945
|
+
* array. Ax binds the field as `findings:json[]` so the provider emits
|
|
3946
|
+
* native structured output; at the kind-factory boundary we Zod-validate
|
|
3947
|
+
* each emitted finding so malformed rows fail loud instead of being
|
|
3948
|
+
* silently lifted with default severity.
|
|
3949
|
+
*
|
|
3950
|
+
* Why not `f.object().array()` directly in the signature? The Ax
|
|
3951
|
+
* signature string `question:string -> findings:json[]` already lets
|
|
3952
|
+
* the provider emit JSON arrays. A Zod boundary is required either
|
|
3953
|
+
* way (the provider can return any JSON), and Zod gives us a single
|
|
3954
|
+
* validation surface independent of which Ax version is installed.
|
|
3955
|
+
*/
|
|
3956
|
+
|
|
3957
|
+
declare const RawAnalystFindingSchema: z.ZodObject<{
|
|
3958
|
+
severity: z.ZodEnum<{
|
|
3959
|
+
low: "low";
|
|
3960
|
+
high: "high";
|
|
3961
|
+
critical: "critical";
|
|
3962
|
+
medium: "medium";
|
|
3963
|
+
info: "info";
|
|
3964
|
+
}>;
|
|
3965
|
+
claim: z.ZodString;
|
|
3966
|
+
subject: z.ZodOptional<z.ZodString>;
|
|
3967
|
+
evidence_uri: z.ZodString;
|
|
3968
|
+
evidence_excerpt: z.ZodOptional<z.ZodString>;
|
|
3969
|
+
confidence: z.ZodNumber;
|
|
3970
|
+
rationale: z.ZodOptional<z.ZodString>;
|
|
3971
|
+
recommended_action: z.ZodOptional<z.ZodString>;
|
|
3972
|
+
}, z.core.$strict>;
|
|
3973
|
+
type RawAnalystFinding = z.infer<typeof RawAnalystFindingSchema>;
|
|
3974
|
+
|
|
3975
|
+
/**
|
|
3976
|
+
* Analyst-kind factory — the typed way to define trace analysts.
|
|
3977
|
+
*
|
|
3978
|
+
* A "kind" is a specialized analyst whose actor prompt, tool subset,
|
|
3979
|
+
* and Ax recursion config target one failure-mode lens (failure-mode
|
|
3980
|
+
* classification, knowledge gap discovery, knowledge poisoning, recursive
|
|
3981
|
+
* self-improvement, ...). Kinds emit findings in the typed `RawAnalystFinding`
|
|
3982
|
+
* shape via a JSON-array Ax output; the factory validates each row with
|
|
3983
|
+
* Zod and lifts it into `AnalystFinding[]` with no shape guessing.
|
|
3984
|
+
*
|
|
3985
|
+
* Composition rules:
|
|
3986
|
+
* - Each kind owns its actor description. No generic "answer this
|
|
3987
|
+
* question" prompt — the prompt names the failure lens.
|
|
3988
|
+
* - Each kind picks a narrow tool subset from `ANALYST_TOOL_GROUPS`.
|
|
3989
|
+
* A kind that never needs full-trace dumps can drop `viewTrace` /
|
|
3990
|
+
* `viewSpans` and stay cheap.
|
|
3991
|
+
* - Each kind declares its recursion + parallelism budget. Discovery-
|
|
3992
|
+
* heavy kinds (failure-mode) get higher `maxDepth`; lens kinds
|
|
3993
|
+
* (poisoning) usually stay at 0 since they have a tighter brief.
|
|
3994
|
+
*
|
|
3995
|
+
* Optimizer hook: kinds may declare `goldens` — labeled examples used
|
|
3996
|
+
* by `AxMiPRO` / `AxBootstrapFewShot` / `AxGEPA` to fit the actor
|
|
3997
|
+
* description programmatically. Stored on the kind, not the registry,
|
|
3998
|
+
* because the right metric is kind-specific.
|
|
3999
|
+
*/
|
|
4000
|
+
|
|
4001
|
+
/**
|
|
4002
|
+
* Per-kind specification. The factory turns this into a regular
|
|
4003
|
+
* `Analyst<TraceAnalysisStore>` ready for `AnalystRegistry.register()`.
|
|
4004
|
+
*/
|
|
4005
|
+
interface TraceAnalystKindSpec {
|
|
4006
|
+
/** Stable id. Appears in finding_id, telemetry, and registry exclusions. */
|
|
4007
|
+
id: string;
|
|
4008
|
+
/** One-sentence description shown in `registry.list()`. */
|
|
4009
|
+
description: string;
|
|
4010
|
+
/** Coarse classification stamped on every emitted finding (`failure-mode`, `knowledge-gap`, ...). */
|
|
4011
|
+
area: string;
|
|
4012
|
+
/** Bump on any breaking change to the actor prompt or output schema. */
|
|
4013
|
+
version: string;
|
|
4014
|
+
/** Actor system prompt. Must instruct the LLM to emit `findings` per the schema. */
|
|
4015
|
+
actorDescription: string;
|
|
4016
|
+
/** Responder system prompt; falls back to a minimal "format the findings" instruction. */
|
|
4017
|
+
responderDescription?: string;
|
|
4018
|
+
/** Tool functions the actor may call. Pick narrow subsets via `ANALYST_TOOL_GROUPS`. */
|
|
4019
|
+
buildTools: (store: TraceAnalysisStore) => AxFunction[];
|
|
4020
|
+
/** Recursion budget. `maxDepth: 0` disables subagents. */
|
|
4021
|
+
recursion?: {
|
|
4022
|
+
maxDepth: number;
|
|
4023
|
+
maxParallelSubagents?: number;
|
|
4024
|
+
};
|
|
4025
|
+
/** Actor turn cap. Default 12. */
|
|
4026
|
+
maxTurns?: number;
|
|
4027
|
+
/** Runtime char cap. Default 6000. */
|
|
4028
|
+
maxRuntimeChars?: number;
|
|
4029
|
+
/** Cost classification surfaced in `registry.list()` and budget enforcement. */
|
|
4030
|
+
cost: AnalystCost;
|
|
4031
|
+
/** Per-finding-row hook — kinds may reject / rewrite before lifting. */
|
|
4032
|
+
postProcess?: (row: RawAnalystFinding, ctx: AnalystContext) => RawAnalystFinding | null;
|
|
4033
|
+
/** Optional optimizer hook — populated when a kind wants to fit its prompt against labeled examples. */
|
|
4034
|
+
goldens?: TraceAnalystGolden[];
|
|
4035
|
+
}
|
|
4036
|
+
/**
|
|
4037
|
+
* One labeled example consumed by Ax optimizers (MIPRO / GEPA / Bootstrap).
|
|
4038
|
+
* Each input is the same `{question}` an analyst would receive; `expected`
|
|
4039
|
+
* is the ground-truth finding set a fitted prompt should produce on this
|
|
4040
|
+
* input. Metric: kind-specific (default: F1 on `finding_id` overlap).
|
|
4041
|
+
*/
|
|
4042
|
+
interface TraceAnalystGolden {
|
|
4043
|
+
question: string;
|
|
4044
|
+
expected: ReadonlyArray<Omit<RawAnalystFinding, 'confidence'>>;
|
|
4045
|
+
}
|
|
4046
|
+
|
|
4047
|
+
/**
|
|
4048
|
+
* AnalystRegistry — orchestrate N analysts against one run.
|
|
4049
|
+
*
|
|
4050
|
+
* Owns three responsibilities and only three:
|
|
4051
|
+
* 1. Registration — ids must be unique; bad registrations fail loudly
|
|
4052
|
+
* at register-time, not run-time.
|
|
4053
|
+
* 2. Routing — each analyst declares its `inputKind`; the registry
|
|
4054
|
+
* picks the matching field from AnalystRunInputs and skips the
|
|
4055
|
+
* analyst with a logged reason if it's missing.
|
|
4056
|
+
* 3. Isolation — one analyst's exception MUST NOT stop other analysts.
|
|
4057
|
+
* Failed analysts produce zero findings + a 'failed' summary row.
|
|
4058
|
+
*
|
|
4059
|
+
* Cross-cutting concerns (telemetry, error → finding conversion, cost
|
|
4060
|
+
* ingestion, storage rotation) live in `AnalystHooks`. Budget shaping
|
|
4061
|
+
* (equal split vs weighted vs custom) lives in `BudgetPolicy`. Both
|
|
4062
|
+
* have sensible defaults; consumers override only what they need.
|
|
4063
|
+
*/
|
|
4064
|
+
|
|
4065
|
+
interface AnalystHooks {
|
|
4066
|
+
/** Before analyze() — last chance to mutate ctx (e.g. inject tags, override budget). */
|
|
4067
|
+
onBeforeAnalyze?(args: {
|
|
4068
|
+
analyst: Analyst;
|
|
4069
|
+
ctx: AnalystContext;
|
|
4070
|
+
runId: string;
|
|
4071
|
+
}): void | Promise<void>;
|
|
4072
|
+
/** After every analyst (ok | failed | skipped). Use for telemetry, ingestion, rotation. */
|
|
4073
|
+
onAfterAnalyze?(args: {
|
|
4074
|
+
analyst: Analyst;
|
|
4075
|
+
summary: AnalystRunSummary;
|
|
4076
|
+
findings: AnalystFinding[];
|
|
4077
|
+
runId: string;
|
|
4078
|
+
}): void | Promise<void>;
|
|
4079
|
+
/**
|
|
4080
|
+
* On analyst exception. Hook MAY return findings to convert the
|
|
4081
|
+
* error into structured findings; the summary still reports 'failed'.
|
|
4082
|
+
* Return void to keep the default empty-findings behavior.
|
|
4083
|
+
*/
|
|
4084
|
+
onError?(args: {
|
|
4085
|
+
analyst: Analyst;
|
|
4086
|
+
error: Error;
|
|
4087
|
+
runId: string;
|
|
4088
|
+
}): AnalystFinding[] | undefined | Promise<AnalystFinding[] | undefined>;
|
|
4089
|
+
/** Once after registry.run() completes. Use for final aggregation, persistence. */
|
|
4090
|
+
onComplete?(args: {
|
|
4091
|
+
result: AnalystRunResult;
|
|
4092
|
+
}): void | Promise<void>;
|
|
4093
|
+
}
|
|
4094
|
+
interface BudgetPolicy {
|
|
4095
|
+
/** Overall USD cap across the registry.run(). */
|
|
4096
|
+
totalUsd?: number;
|
|
4097
|
+
/** Per-analyst weight for the default allocator. Missing ids get weight 1. */
|
|
4098
|
+
weights?: Record<string, number>;
|
|
4099
|
+
/**
|
|
4100
|
+
* Custom allocator — receives the analyst, remaining/total budget, and
|
|
4101
|
+
* the count of analysts that will run. Returns the per-analyst budget
|
|
4102
|
+
* (or undefined to leave it uncapped). Overrides weights when set.
|
|
4103
|
+
*/
|
|
4104
|
+
allocate?: (args: {
|
|
4105
|
+
analyst: Analyst;
|
|
4106
|
+
totalUsd: number | undefined;
|
|
4107
|
+
remainingUsd: number | undefined;
|
|
4108
|
+
runningCount: number;
|
|
4109
|
+
}) => number | undefined;
|
|
4110
|
+
}
|
|
4111
|
+
interface AnalystRegistryOptions {
|
|
4112
|
+
/** Shared chat client passed to every LLM analyst via AnalystContext. */
|
|
4113
|
+
chat?: ChatClient;
|
|
4114
|
+
/** Logger callback. Defaults to a no-op. */
|
|
4115
|
+
log?: (msg: string, fields?: Record<string, unknown>) => void;
|
|
4116
|
+
/** Hooks invoked around analyze() — observability + customization seam. */
|
|
4117
|
+
hooks?: AnalystHooks;
|
|
4118
|
+
/** Default budget when run() doesn't override. */
|
|
4119
|
+
defaultBudget?: BudgetPolicy;
|
|
4120
|
+
}
|
|
4121
|
+
interface RegistryRunOpts {
|
|
4122
|
+
/** Restrict to a subset of registered analysts by id. */
|
|
4123
|
+
only?: string[];
|
|
4124
|
+
/** Skip these analysts even if registered. Useful for cheap iteration. */
|
|
4125
|
+
skip?: string[];
|
|
4126
|
+
/** Budget policy — totalUsd + optional weights/allocator. Falls back to options.defaultBudget. */
|
|
4127
|
+
budget?: BudgetPolicy;
|
|
4128
|
+
/** Wall-clock cap. Analysts SHOULD honor `ctx.deadlineMs`. */
|
|
4129
|
+
timeoutMs?: number;
|
|
4130
|
+
/** Abort signal — forwarded into every analyst's context. */
|
|
4131
|
+
signal?: AbortSignal;
|
|
4132
|
+
/** Tags echoed into AnalystContext.tags — useful for tracking environment/version in findings. */
|
|
4133
|
+
tags?: Record<string, string>;
|
|
4134
|
+
/**
|
|
4135
|
+
* Prior-run findings made available as retrieval context to every
|
|
4136
|
+
* analyst via `ctx.priorFindings`. The registry forwards the slice
|
|
4137
|
+
* whose `analyst_id` matches each registered analyst so a kind sees
|
|
4138
|
+
* only its own history. Pass `{ '*': findings }` to broadcast to
|
|
4139
|
+
* every analyst (useful for cross-kind chaining where the improvement
|
|
4140
|
+
* analyst consumes upstream failure findings).
|
|
4141
|
+
*/
|
|
4142
|
+
priorFindings?: ReadonlyArray<AnalystFinding> | Record<string, ReadonlyArray<AnalystFinding>>;
|
|
4143
|
+
}
|
|
4144
|
+
declare class AnalystRegistry {
|
|
4145
|
+
private readonly analysts;
|
|
4146
|
+
private readonly options;
|
|
4147
|
+
constructor(options?: AnalystRegistryOptions);
|
|
4148
|
+
register(analyst: Analyst): void;
|
|
4149
|
+
list(): ReadonlyArray<{
|
|
4150
|
+
id: string;
|
|
4151
|
+
description: string;
|
|
4152
|
+
version: string;
|
|
4153
|
+
cost: Analyst['cost'];
|
|
4154
|
+
}>;
|
|
4155
|
+
run(runId: string, inputs: AnalystRunInputs, runOpts?: RegistryRunOpts): Promise<AnalystRunResult>;
|
|
4156
|
+
/**
|
|
4157
|
+
* Streaming counterpart to `run()`. Emits `AnalystRunEvent` values
|
|
4158
|
+
* in real time — `run-started`, then per-analyst `skipped` /
|
|
4159
|
+
* `started` / `completed`, then a terminal `run-completed` whose
|
|
4160
|
+
* payload is the full `AnalystRunResult`. UIs use this to render
|
|
4161
|
+
* progress; persistence consumers use `run()` and read the result.
|
|
4162
|
+
*
|
|
4163
|
+
* Hooks (`onBeforeAnalyze` / `onAfterAnalyze` / `onError` /
|
|
4164
|
+
* `onComplete`) fire as before — streaming is additive, not a hook
|
|
4165
|
+
* replacement.
|
|
4166
|
+
*/
|
|
4167
|
+
runStream(runId: string, inputs: AnalystRunInputs, runOpts?: RegistryRunOpts): AsyncGenerator<AnalystRunEvent, void, void>;
|
|
4168
|
+
private selectAnalysts;
|
|
4169
|
+
private routeInput;
|
|
4170
|
+
}
|
|
4171
|
+
|
|
4172
|
+
/**
|
|
4173
|
+
* `buildDefaultAnalystRegistry` — the canonical analyst suite, so consumers
|
|
4174
|
+
* stop hand-wiring `new AnalystRegistry()` + per-kind `createTraceAnalystKind`.
|
|
4175
|
+
*
|
|
4176
|
+
* The deterministic `behavioralAnalyst` is ALWAYS registered (it needs no
|
|
4177
|
+
* model and is model-agnostic by construction). The agentic RLM kinds are
|
|
4178
|
+
* registered only when an `ai` service is supplied — so a caller with no LLM
|
|
4179
|
+
* still gets the full behavioral/efficiency diagnosis, and the substrate's
|
|
4180
|
+
* "any model (including no model)" guarantee holds at the suite level.
|
|
4181
|
+
*/
|
|
4182
|
+
|
|
4183
|
+
interface DefaultAnalystRegistryOptions {
|
|
4184
|
+
/** Ax service for the agentic RLM kinds. Omit → only the deterministic analyst. */
|
|
4185
|
+
ai?: AxAIService;
|
|
4186
|
+
/** Model for the agentic kinds (falls back to the ai service default). */
|
|
4187
|
+
model?: string;
|
|
4188
|
+
/** Which agentic kinds to register when `ai` is present. Default = the shipped suite. */
|
|
4189
|
+
kinds?: readonly TraceAnalystKindSpec[];
|
|
4190
|
+
/** Set false to omit the deterministic behavioral analyst (default: include). */
|
|
4191
|
+
includeBehavioral?: boolean;
|
|
4192
|
+
/** Forwarded to the AnalystRegistry constructor (signal, tags, priorFindings). */
|
|
4193
|
+
registry?: AnalystRegistryOptions;
|
|
4194
|
+
}
|
|
4195
|
+
declare function buildDefaultAnalystRegistry(opts?: DefaultAnalystRegistryOptions): AnalystRegistry;
|
|
4196
|
+
|
|
4197
|
+
/**
|
|
4198
|
+
* # `analyzeRuns()` — turn a set of agent runs into an actionable decision packet.
|
|
4199
|
+
*
|
|
4200
|
+
* Wires the substrate's statistical, calibration, clustering, Pareto, and
|
|
4201
|
+
* release-confidence primitives into one `InsightReport`. Two top-level
|
|
4202
|
+
* entry points use this function:
|
|
4203
|
+
*
|
|
4204
|
+
* - `selfImprove()` calls it on the campaign output to attach a packet
|
|
4205
|
+
* to every run.
|
|
4206
|
+
* - Consumers with observed `RunRecord[]` (production traces, gold
|
|
4207
|
+
* corpora, approve/reject tables) call it directly via `analyzeRuns()`
|
|
4208
|
+
* for analysis without a closed loop.
|
|
4209
|
+
*
|
|
4210
|
+
* Every section is opt-in based on what the input data supports — the
|
|
4211
|
+
* function never invents signal. If runs carry no judge scores, `judges`
|
|
4212
|
+
* is empty. If there's no baseline/candidate split, `lift` is undefined.
|
|
4213
|
+
* If no `analyst` is wired, `failureClusters` is undefined.
|
|
4214
|
+
*
|
|
4215
|
+
* The `recommendations` array is the human-readable layer; everything
|
|
4216
|
+
* else is the evidence backing each recommendation.
|
|
4217
|
+
*/
|
|
4218
|
+
|
|
4219
|
+
interface AnalyzeRunsOptions {
|
|
4220
|
+
/** The runs to analyze. */
|
|
4221
|
+
runs: RunRecord[];
|
|
4222
|
+
/** Which split to score against when reading composite from RunOutcome.
|
|
4223
|
+
* Default: holdout when ANY run has a `holdoutScore`, else search. */
|
|
4224
|
+
split?: 'search' | 'holdout' | 'auto';
|
|
4225
|
+
/** Pairwise analysis configuration. When both `baselineCandidateId` and
|
|
4226
|
+
* `candidateCandidateId` are present, lift is computed on paired
|
|
4227
|
+
* (experimentId, seed) tuples shared between the two sides. */
|
|
4228
|
+
baselineCandidateId?: string;
|
|
4229
|
+
candidateCandidateId?: string;
|
|
4230
|
+
/** Canary scenarios — checked against every run's raw output for
|
|
4231
|
+
* holdout contamination. */
|
|
4232
|
+
canaryScenarios?: DatasetScenario[];
|
|
4233
|
+
/** Analyst registry for failure clustering. When omitted, the
|
|
4234
|
+
* `failureClusters` section is left undefined. */
|
|
4235
|
+
analyst?: AnalystRegistry;
|
|
4236
|
+
/** Downstream outcome metric per run (e.g. engagement rate, approval
|
|
4237
|
+
* rate, downstream pass rate). When present, the report includes
|
|
4238
|
+
* `outcomeCorrelation` + a simple linear reward model fit. */
|
|
4239
|
+
outcomeSignal?: {
|
|
4240
|
+
metric: string;
|
|
4241
|
+
valueByRunId: Record<string, number>;
|
|
4242
|
+
};
|
|
4243
|
+
/** Multi-rater feedback for inter-rater agreement. Each entry is one
|
|
4244
|
+
* rater's score for one run. Two or more raters → kappa + disagreement
|
|
4245
|
+
* triage list. */
|
|
4246
|
+
raterScores?: Array<{
|
|
4247
|
+
runId: string;
|
|
4248
|
+
rater: string;
|
|
4249
|
+
score: number;
|
|
4250
|
+
}>;
|
|
4251
|
+
/** Number of histogram bins for distributional summaries. Default 12. */
|
|
4252
|
+
histogramBins?: number;
|
|
4253
|
+
/** Decision threshold — the smallest composite lift the caller cares
|
|
4254
|
+
* about. Used by the recommendations engine to call ship vs hold.
|
|
4255
|
+
* Default 0.02. */
|
|
4256
|
+
decisionThreshold?: number;
|
|
4257
|
+
/** Optional prior-period runs. When set, the report includes
|
|
4258
|
+
* `priorPeriodComparison` with per-metric Welch-CI deltas and
|
|
4259
|
+
* recommendations fire on statistically significant regressions.
|
|
4260
|
+
* The two windows do NOT have to share scenarios — the comparison
|
|
4261
|
+
* is two-sample unpaired (the substrate's `lift` field uses paired
|
|
4262
|
+
* bootstrap on shared (experimentId, seed) tuples; this is the
|
|
4263
|
+
* shape for "this week vs last week" rather than "candidate vs
|
|
4264
|
+
* baseline within a campaign"). */
|
|
4265
|
+
baselineRuns?: RunRecord[];
|
|
4266
|
+
/** Human-readable label for the baseline window, e.g. "vs prior 7
|
|
4267
|
+
* days", "vs v3.1 release". Surfaces in recommendations + UI. */
|
|
4268
|
+
baselineLabel?: string;
|
|
4269
|
+
}
|
|
4270
|
+
interface SummarizeExecutionOptions {
|
|
4271
|
+
runs: RunRecord[];
|
|
4272
|
+
histogramBins?: number;
|
|
4273
|
+
}
|
|
4274
|
+
interface ExecutionReport {
|
|
4275
|
+
execution: ExecutionInsight;
|
|
4276
|
+
costProvenance: CostProvenanceSummary;
|
|
4277
|
+
}
|
|
4278
|
+
/** Summarize runtime facts without interpreting task quality or promotion readiness. */
|
|
4279
|
+
declare function summarizeExecution(opts: SummarizeExecutionOptions): ExecutionReport;
|
|
4280
|
+
declare function analyzeRuns(opts: AnalyzeRunsOptions): Promise<InsightReport>;
|
|
4281
|
+
|
|
395
4282
|
/**
|
|
396
4283
|
* # `intake/run-record-dir` — load a directory or file of `RunRecord`s.
|
|
397
4284
|
*
|
|
@@ -736,6 +4623,92 @@ interface PartitionByAuthoringModelResult {
|
|
|
736
4623
|
*/
|
|
737
4624
|
declare function partitionRunsByAuthoringModel(runs: RunRecord[], index: AgentTraceIndex): PartitionByAuthoringModelResult;
|
|
738
4625
|
|
|
4626
|
+
type CodeAgentSessionSource = 'codex' | 'claude-code' | 'opencode' | 'kimi-code' | 'pi';
|
|
4627
|
+
interface ParsedCodeAgentJsonl {
|
|
4628
|
+
entries: unknown[];
|
|
4629
|
+
malformedLines: number;
|
|
4630
|
+
}
|
|
4631
|
+
interface CodeAgentSessionMetrics {
|
|
4632
|
+
entries: number;
|
|
4633
|
+
userMessages: number;
|
|
4634
|
+
assistantMessages: number;
|
|
4635
|
+
reasoningItems: number;
|
|
4636
|
+
toolCalls: number;
|
|
4637
|
+
toolOutputs: number;
|
|
4638
|
+
toolErrors: number;
|
|
4639
|
+
patchAttempts: number;
|
|
4640
|
+
patchSuccesses: number;
|
|
4641
|
+
patchFailures: number;
|
|
4642
|
+
turnsStarted: number;
|
|
4643
|
+
turnsCompleted: number;
|
|
4644
|
+
turnsAborted: number;
|
|
4645
|
+
contextCompactions: number;
|
|
4646
|
+
prLinks: number;
|
|
4647
|
+
fileSnapshots: number;
|
|
4648
|
+
graphNodes: number;
|
|
4649
|
+
graphEdges: number;
|
|
4650
|
+
actionCandidates: number;
|
|
4651
|
+
verificationReports: number;
|
|
4652
|
+
completionDecisions: number;
|
|
4653
|
+
reliabilityRows: number;
|
|
4654
|
+
reliabilityLift: number;
|
|
4655
|
+
inputTokens: number;
|
|
4656
|
+
outputTokens: number;
|
|
4657
|
+
reasoningTokens: number;
|
|
4658
|
+
cachedTokens: number;
|
|
4659
|
+
cacheWriteTokens: number;
|
|
4660
|
+
observedCostUsd: number;
|
|
4661
|
+
observedCostCaptured?: boolean;
|
|
4662
|
+
wallMs: number;
|
|
4663
|
+
processScore: number;
|
|
4664
|
+
}
|
|
4665
|
+
interface CodeAgentSessionDiagnostic {
|
|
4666
|
+
source: CodeAgentSessionSource;
|
|
4667
|
+
sessionId: string;
|
|
4668
|
+
sourcePath?: string;
|
|
4669
|
+
entries: number;
|
|
4670
|
+
malformedLines: number;
|
|
4671
|
+
inferredScore: boolean;
|
|
4672
|
+
hasExplicitTerminalSignal: boolean;
|
|
4673
|
+
hasQualityLabel: boolean;
|
|
4674
|
+
hasTokenUsage: boolean;
|
|
4675
|
+
hasCost: boolean;
|
|
4676
|
+
costKind?: RunCostProvenance['kind'];
|
|
4677
|
+
warnings: string[];
|
|
4678
|
+
}
|
|
4679
|
+
interface CodeAgentSessionIntakeResult {
|
|
4680
|
+
runs: RunRecord[];
|
|
4681
|
+
diagnostics: CodeAgentSessionDiagnostic[];
|
|
4682
|
+
metrics: CodeAgentSessionMetrics[];
|
|
4683
|
+
}
|
|
4684
|
+
interface CodeAgentSessionIntakeOptions {
|
|
4685
|
+
entries: unknown[];
|
|
4686
|
+
malformedLines?: number;
|
|
4687
|
+
sourcePath?: string;
|
|
4688
|
+
experimentId?: string;
|
|
4689
|
+
candidateId?: string;
|
|
4690
|
+
seed?: number;
|
|
4691
|
+
splitTag?: RunSplitTag;
|
|
4692
|
+
scenarioId?: string;
|
|
4693
|
+
model?: string;
|
|
4694
|
+
promptHash?: string;
|
|
4695
|
+
configHash?: string;
|
|
4696
|
+
commitSha?: string;
|
|
4697
|
+
score?: number;
|
|
4698
|
+
/** Explicit cost receipt. Use `uncaptured` when the source says dollars
|
|
4699
|
+
* were not captured; the adapter will not relabel its compatibility $0
|
|
4700
|
+
* sentinel as observed. When omitted, source-reported cost wins, then a
|
|
4701
|
+
* token-priced estimate, then uncaptured. */
|
|
4702
|
+
costProvenance?: RunCostProvenance;
|
|
4703
|
+
}
|
|
4704
|
+
declare function parseCodeAgentJsonl(jsonl: string): ParsedCodeAgentJsonl;
|
|
4705
|
+
declare function fromCodexSession(options: CodeAgentSessionIntakeOptions): CodeAgentSessionIntakeResult;
|
|
4706
|
+
declare function fromClaudeCodeSession(options: CodeAgentSessionIntakeOptions): CodeAgentSessionIntakeResult;
|
|
4707
|
+
declare function fromOpenCodeSession(options: CodeAgentSessionIntakeOptions): CodeAgentSessionIntakeResult;
|
|
4708
|
+
declare function fromKimiCodeSession(options: CodeAgentSessionIntakeOptions): CodeAgentSessionIntakeResult;
|
|
4709
|
+
declare function fromPiSession(options: CodeAgentSessionIntakeOptions): CodeAgentSessionIntakeResult;
|
|
4710
|
+
declare const fromPigraphSession: typeof fromPiSession;
|
|
4711
|
+
|
|
739
4712
|
/**
|
|
740
4713
|
* # `intake/feedback-table` — multi-rater approve/reject corpus → `RunRecord[]`.
|
|
741
4714
|
*
|
|
@@ -838,7 +4811,8 @@ declare function fromFeedbackTable(opts: FromFeedbackTableOptions): FromFeedback
|
|
|
838
4811
|
* - `wallMs` from `endTimeUnixNano - startTimeUnixNano`
|
|
839
4812
|
* - `model` from `gen_ai.request.model` / `llm.model` / `tangle.model`
|
|
840
4813
|
* - cost from `cost.usd` / `gen_ai.usage.cost_usd` / `tangle.cost.usd`
|
|
841
|
-
* - token usage from
|
|
4814
|
+
* - token usage from model-call input, output, cache-read, and cache-write
|
|
4815
|
+
* attributes without double-counting aggregate parent spans
|
|
842
4816
|
* - `outcome.searchScore` from `tangle.score` / `eval.score` when
|
|
843
4817
|
* present; `outcome.raw` collects every numeric attribute.
|
|
844
4818
|
*
|
|
@@ -855,4 +4829,4 @@ interface FromOtelSpansOptions {
|
|
|
855
4829
|
}
|
|
856
4830
|
declare function fromOtelSpans(opts: FromOtelSpansOptions): RunRecord[];
|
|
857
4831
|
|
|
858
|
-
export { type AgentEvalAgent, type AgentEvalEvaluateOptions, type AgentEvalImproveOptions, type AgentTraceContributor, type AgentTraceContributorType, type AgentTraceConversation, type AgentTraceFile, type AgentTraceIndex, type AgentTraceRange, type AgentTraceRecord, AnalyzeRunsOptions, type AuthoringProvenance, CampaignResult, CampaignStorage, type DefineAgentEvalOptions, type DefinedAgentEval, DispatchContext, type EvalCellScoreDelta, type EvalDimensionDelta, type EvalGenerationDiff, type EvalReportingSuiteInput, type EvalReportingSuiteOptions, type EvalReportingSuiteResult, type EvalRunDiff, type FeedbackTableMeta, type FeedbackTableRow, type FromFeedbackTableOptions, type FromFeedbackTableResult, type FromOtelSpansOptions, type FromRunRecordDirOptions, type FromRunRecordDirResult, Gate, GateDecision, HostedTenant, InsightReport, JudgeConfig, MutableSurface, type PartitionByAuthoringModelResult, RunEvalOptions, RunImprovementLoopResult, type RunRecordRejection, Scenario, type SelfImproveBudget, type SelfImproveLlm, type SelfImproveOptions, type SelfImproveProgressEvent, type SelfImproveResult, SelfImproveRunError, SurfaceProposer, defineAgentEval, diffGenerations, diffRunBaselineToWinner, diffRuns, evalReportingSuite, fromFeedbackTable, fromOtelSpans, fromRunRecordDir, parseAgentTrace, partitionRunsByAuthoringModel, selfImprove };
|
|
4832
|
+
export { type AgentEvalAgent, type AgentEvalEvaluateOptions, type AgentEvalImproveOptions, type AgentTraceContributor, type AgentTraceContributorType, type AgentTraceConversation, type AgentTraceFile, type AgentTraceIndex, type AgentTraceRange, type AgentTraceRecord, type AnalystFinding, type AnalyzeRunsOptions, type AuthoringProvenance, type AxisEvidence, type AxisVerdict, type BuildEvidenceVectorOptions, type CampaignAggregates, type CampaignArtifactWriter, type CampaignCellResult, type CampaignCostMeter, type CampaignResult, type CampaignStorage, type CampaignTraceWriter, type ChatClient, type CodeAgentSessionDiagnostic, type CodeAgentSessionIntakeOptions, type CodeAgentSessionIntakeResult, type CodeAgentSessionMetrics, type CodeAgentSessionSource, type CodeSurface, type CostProvenanceSummary, type CreateChatClientOpts, type DefaultAnalystRegistryOptions, type DefaultProductionGateOptions, type DefineAgentEvalOptions, type DefinedAgentEval, type DeploymentOutcome, type DispatchFn as Dispatch, type DispatchContext, type EvalCellScoreDelta, type EvalDimensionDelta, type EvalGenerationDiff, type EvalReportingSuiteInput, type EvalReportingSuiteOptions, type EvalReportingSuiteResult, type EvalRunDiff, type EvidenceVector, type EvolutionaryProposerOptions, type ExecutionInsight, type ExecutionReport, type FailureClusterInsight, type FeedbackTableMeta, type FeedbackTableRow, FileSystemOutcomeStore, type FileSystemOutcomeStoreOptions, type FromFeedbackTableOptions, type FromFeedbackTableResult, type FromOtelSpansOptions, type FromRunRecordDirOptions, type FromRunRecordDirResult, type Gate, type GateContext, type GateDecision, type GateResult, type GenerationCandidate, type GenerationRecord, type GepaProposerOptions, type HeldOutGateOptions, type HostedTenant, InMemoryOutcomeStore, type InsightReport, type InterRaterInsight, type JudgeConfig, type JudgeDimension, type JudgeInsight, type JudgeScore, type LiftInsight, type MutableSurface, type Mutator, type ObjectiveSource, type OptimizationProposer, type OptimizerConfig, type OutcomeCorrelationInsight, type OutcomeStore, type ParetoSignificanceGateOptions, type ParsedCodeAgentJsonl, type PartitionByAuthoringModelResult, type PromotionObjective, type PromotionPolicy, REFERENCE_EQUIVALENCE_INPUT_LIMITS, REFERENCE_EQUIVALENCE_JUDGE_VERSION, type Recommendation, type ReferenceEquivalenceJudgeInput, type ReferenceEquivalenceJudgeOptions, type ReferenceEquivalenceJudgeResult, type ReferenceEquivalenceScenario, type ReleaseSummary, type RunCampaignOptions, type RunEvalOptions, type RunImprovementLoopOptions, type RunImprovementLoopResult, type RunRecordRejection, type ScalarDistribution, type Scenario, type SelfImproveBudget, type SelfImproveLlm, type SelfImproveOptions, type SelfImproveProgressEvent, type SelfImproveResult, SelfImproveRunError, type SessionScript, type SummarizeExecutionOptions, type SurfaceProposer, type TokenUsageInsight, analyzeRuns, buildDefaultAnalystRegistry, buildEvidenceVector, composeGate, createChatClient, createReferenceEquivalenceJudge, defaultProductionGate, defineAgentEval, diffGenerations, diffRunBaselineToWinner, diffRuns, evalReportingSuite, evolutionaryProposer, fromClaudeCodeSession, fromCodexSession, fromFeedbackTable, fromKimiCodeSession, fromOpenCodeSession, fromOtelSpans, fromPiSession, fromPigraphSession, fromRunRecordDir, fsCampaignStorage, gepaProposer, heldOutGate, inMemoryCampaignStorage, paretoPolicy, paretoSignificanceGate, parseAgentTrace, parseCodeAgentJsonl, partitionRunsByAuthoringModel, runCampaign, runEval, runImprovementLoop, runReferenceEquivalenceJudge, selfImprove, summarizeExecution };
|