@tangle-network/agent-eval 0.117.1 → 0.118.1
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +30 -0
- package/dist/analyst/index.d.ts +2772 -21
- package/dist/analyst/index.js +7 -6
- package/dist/analyst/index.js.map +1 -1
- package/dist/belief-state/index.d.ts +706 -10
- package/dist/belief-state/index.js +2 -1
- package/dist/belief-state/index.js.map +1 -1
- package/dist/benchmarks/index.d.ts +958 -14
- package/dist/benchmarks/index.js +12 -10
- package/dist/builder-eval/index.d.ts +449 -4
- package/dist/builder-eval/index.js +4 -3
- package/dist/builder-eval/index.js.map +1 -1
- package/dist/campaign/index.d.ts +4275 -73
- package/dist/campaign/index.js +12 -10
- package/dist/{chunk-VF3XSYTI.js → chunk-33JA4TFA.js} +6 -6
- package/dist/{chunk-4JLWXDYA.js → chunk-3EHHMC6E.js} +2 -2
- package/dist/{chunk-CCZIVI3F.js → chunk-BFW56GTT.js} +2 -2
- package/dist/{chunk-YZPO4UHR.js → chunk-FTUMG2U7.js} +124 -149
- package/dist/chunk-FTUMG2U7.js.map +1 -0
- package/dist/{chunk-E4BUPP7Z.js → chunk-HKUCJ437.js} +38 -66
- package/dist/chunk-HKUCJ437.js.map +1 -0
- package/dist/{chunk-HZHNRYHK.js → chunk-K6N6XJJX.js} +2 -2
- package/dist/chunk-KSDQVPLR.js +286 -0
- package/dist/chunk-KSDQVPLR.js.map +1 -0
- package/dist/chunk-MA6HLL3S.js +65 -0
- package/dist/chunk-MA6HLL3S.js.map +1 -0
- package/dist/{chunk-MGEHEHSN.js → chunk-OIIMMLRB.js} +11 -11
- package/dist/{chunk-DXZRATT5.js → chunk-OYZAPX5G.js} +3 -3
- package/dist/chunk-PXE2VKMX.js +140 -0
- package/dist/chunk-PXE2VKMX.js.map +1 -0
- package/dist/{chunk-JSJZ4PJ6.js → chunk-Q442S5AS.js} +17 -17
- package/dist/{chunk-ODVOOEWQ.js → chunk-QBRSJK47.js} +2 -2
- package/dist/{chunk-S2F4J57L.js → chunk-QKEGNI5B.js} +77 -32
- package/dist/chunk-QKEGNI5B.js.map +1 -0
- package/dist/{chunk-5UF54T55.js → chunk-S3UZOQ5Y.js} +34 -7
- package/dist/chunk-S3UZOQ5Y.js.map +1 -0
- package/dist/{chunk-FQNLDL4D.js → chunk-SVH2ANFD.js} +136 -4
- package/dist/chunk-SVH2ANFD.js.map +1 -0
- package/dist/{chunk-GQCZRZ7L.js → chunk-U5CHZ5M3.js} +9 -9
- package/dist/{chunk-TVVP3ZZQ.js → chunk-VQMK5FMP.js} +2 -1
- package/dist/{chunk-TVVP3ZZQ.js.map → chunk-VQMK5FMP.js.map} +1 -1
- package/dist/chunk-WDHBCA3M.js +31 -0
- package/dist/chunk-WDHBCA3M.js.map +1 -0
- package/dist/{chunk-HQPHZGL6.js → chunk-YLMUS4MM.js} +9 -9
- package/dist/{chunk-LQUTGLOZ.js → chunk-ZET2UAYW.js} +15 -65
- package/dist/chunk-ZET2UAYW.js.map +1 -0
- package/dist/contract/index.d.ts +4012 -38
- package/dist/contract/index.js +56 -29
- package/dist/contract/index.js.map +1 -1
- package/dist/control.d.ts +1013 -9
- package/dist/control.js +4 -3
- package/dist/fuzz.d.ts +194 -4
- package/dist/fuzz.js +3 -3
- package/dist/hosted/index.d.ts +498 -17
- package/dist/index.d.ts +10992 -1288
- package/dist/index.js +101 -80
- package/dist/index.js.map +1 -1
- package/dist/matrix/index.d.ts +139 -4
- package/dist/meta-eval/index.d.ts +862 -15
- package/dist/meta-eval/index.js +2 -1
- package/dist/meta-eval/index.js.map +1 -1
- package/dist/multishot/index.d.ts +214 -14
- package/dist/openapi.json +1 -1
- package/dist/pipelines/index.d.ts +392 -7
- package/dist/pipelines/index.js +5 -3
- package/dist/pipelines/index.js.map +1 -1
- package/dist/reporting.d.ts +1277 -17
- package/dist/rl.d.ts +2359 -28
- package/dist/rl.js +6 -5
- package/dist/rl.js.map +1 -1
- package/dist/storyboard/index.d.ts +86 -1
- package/dist/trace-attributes.d.ts +16 -0
- package/dist/trace-attributes.js +32 -0
- package/dist/trace-attributes.js.map +1 -0
- package/dist/traces.d.ts +1978 -697
- package/dist/traces.js +53 -32
- package/dist/wire/index.d.ts +655 -9
- package/docs/insight-report.md +44 -0
- package/package.json +9 -3
- package/dist/adversarial-B7loGVVX.d.ts +0 -19
- package/dist/analyst-C8HHvfJp.d.ts +0 -88
- package/dist/analyze-runs--2x39HZ7.d.ts +0 -81
- package/dist/baseline-DKq3gJpP.d.ts +0 -141
- package/dist/calibration-C8MTS7cw.d.ts +0 -101
- package/dist/chunk-5UF54T55.js.map +0 -1
- package/dist/chunk-E4BUPP7Z.js.map +0 -1
- package/dist/chunk-FQNLDL4D.js.map +0 -1
- package/dist/chunk-LQUTGLOZ.js.map +0 -1
- package/dist/chunk-S2F4J57L.js.map +0 -1
- package/dist/chunk-YZPO4UHR.js.map +0 -1
- package/dist/code-agent-session-CjZsVd19.d.ts +0 -87
- package/dist/control-6vuGfmDH.d.ts +0 -258
- package/dist/cost-ledger-DWy3XdJc.d.ts +0 -183
- package/dist/dataset-NENEzRgk.d.ts +0 -115
- package/dist/default-registry-DaK8b3fv.d.ts +0 -155
- package/dist/emitter-CjD7vUwv.d.ts +0 -122
- package/dist/errors-oeQrLqXC.d.ts +0 -74
- package/dist/failure-cluster-DOAcSJ87.d.ts +0 -76
- package/dist/feedback-trajectory-BUnM58xL.d.ts +0 -348
- package/dist/gepa-eESocoDi.d.ts +0 -642
- package/dist/index-PdX4VnPA.d.ts +0 -423
- package/dist/insight-report-DY4nDW9Q.d.ts +0 -310
- package/dist/integrity-DqlBiLyK.d.ts +0 -81
- package/dist/judge-calibration-7C-IDmKr.d.ts +0 -145
- package/dist/kind-factory-ClZmO25A.d.ts +0 -171
- package/dist/llm-client-qoDd18Qz.d.ts +0 -289
- package/dist/multi-layer-verifier-BsqKuLyN.d.ts +0 -150
- package/dist/off-policy-DiwuKKg7.d.ts +0 -132
- package/dist/outcome-store-rnXLEqSn.d.ts +0 -63
- package/dist/policy-edit-wG9uFEFm.d.ts +0 -455
- package/dist/pre-registration-BWQhJ3vz.d.ts +0 -761
- package/dist/provenance-DpjwyseI.d.ts +0 -541
- package/dist/query-CF7PG61p.d.ts +0 -35
- package/dist/raw-provider-sink-C46HDghv.d.ts +0 -132
- package/dist/release-report-C8G2i5Xi.d.ts +0 -236
- package/dist/researcher-C8XyxQsu.d.ts +0 -387
- package/dist/rubric-predictive-validity-p49lLVrE.d.ts +0 -105
- package/dist/run-record-BDH49H2E.d.ts +0 -360
- package/dist/runtime-trajectory-DGBIUt4B.d.ts +0 -49
- package/dist/schema-B3Q3l9Z_.d.ts +0 -201
- package/dist/semantic-concept-judge-CXnPEJbf.d.ts +0 -723
- package/dist/sequential-5iSVfzl2.d.ts +0 -139
- package/dist/series-convergence-D5OWMBg6.d.ts +0 -33
- package/dist/statistics-KUnG73jH.d.ts +0 -494
- package/dist/storage-DrX3v_5B.d.ts +0 -50
- package/dist/store-C1YxJDEK.d.ts +0 -248
- package/dist/store-DGqD0Pyo.d.ts +0 -116
- package/dist/summary-report-C5bKFfm-.d.ts +0 -445
- package/dist/test-graded-scenario-B0ybnPY7.d.ts +0 -166
- package/dist/types-BSw1rOUB.d.ts +0 -634
- package/dist/types-BUxNaJ8c.d.ts +0 -108
- package/dist/types-BkfcQnxV.d.ts +0 -313
- package/dist/verdict-C9MlYujm.d.ts +0 -35
- /package/dist/{chunk-VF3XSYTI.js.map → chunk-33JA4TFA.js.map} +0 -0
- /package/dist/{chunk-4JLWXDYA.js.map → chunk-3EHHMC6E.js.map} +0 -0
- /package/dist/{chunk-CCZIVI3F.js.map → chunk-BFW56GTT.js.map} +0 -0
- /package/dist/{chunk-HZHNRYHK.js.map → chunk-K6N6XJJX.js.map} +0 -0
- /package/dist/{chunk-MGEHEHSN.js.map → chunk-OIIMMLRB.js.map} +0 -0
- /package/dist/{chunk-DXZRATT5.js.map → chunk-OYZAPX5G.js.map} +0 -0
- /package/dist/{chunk-JSJZ4PJ6.js.map → chunk-Q442S5AS.js.map} +0 -0
- /package/dist/{chunk-ODVOOEWQ.js.map → chunk-QBRSJK47.js.map} +0 -0
- /package/dist/{chunk-GQCZRZ7L.js.map → chunk-U5CHZ5M3.js.map} +0 -0
- /package/dist/{chunk-HQPHZGL6.js.map → chunk-YLMUS4MM.js.map} +0 -0
package/dist/gepa-eESocoDi.d.ts
DELETED
|
@@ -1,642 +0,0 @@
|
|
|
1
|
-
import { D as DatasetScenario, c as Dataset } from './dataset-NENEzRgk.js';
|
|
2
|
-
import { T as TraceStore } from './store-DGqD0Pyo.js';
|
|
3
|
-
import { C as ChatClient } from './policy-edit-wG9uFEFm.js';
|
|
4
|
-
import { S as Scenario, b as JudgeConfig, C as CampaignResult, m as GateResult, k as DispatchFn, L as LabeledScenarioStore, i as CampaignTraceWriter, o as GenerationRecord, M as MutableSurface, P as ParetoParent, c as SurfaceProposer, G as Gate } from './types-BSw1rOUB.js';
|
|
5
|
-
import { C as CostLedger, b as CostLedgerSummary } from './cost-ledger-DWy3XdJc.js';
|
|
6
|
-
import { L as LlmCallMetadata, a as LlmClientOptions } from './llm-client-qoDd18Qz.js';
|
|
7
|
-
import { C as CampaignStorage } from './storage-DrX3v_5B.js';
|
|
8
|
-
|
|
9
|
-
/**
|
|
10
|
-
* Pareto frontier — multi-objective optimization over candidate runs.
|
|
11
|
-
*
|
|
12
|
-
* Lifted from ADC pareto.ts and blueprint-agent frontier.ts. When you're
|
|
13
|
-
* trading off (cost, latency, quality) or (passRate, tokenBudget,
|
|
14
|
-
* ttfb), you rarely have a single "winner" — you have a set of
|
|
15
|
-
* non-dominated candidates. This module exposes:
|
|
16
|
-
*
|
|
17
|
-
* - `paretoFrontier`: filter a set of candidates to the non-dominated ones
|
|
18
|
-
* - `dominates`: does A dominate B across all objectives?
|
|
19
|
-
*
|
|
20
|
-
* Each objective is declared with a direction: 'maximize' (higher=better)
|
|
21
|
-
* or 'minimize' (lower=better). Candidates are any object; pass an
|
|
22
|
-
* `objective(candidate)` accessor.
|
|
23
|
-
*/
|
|
24
|
-
type Direction = 'maximize' | 'minimize';
|
|
25
|
-
interface Objective<T> {
|
|
26
|
-
/** Stable label used in reports. */
|
|
27
|
-
name: string;
|
|
28
|
-
direction: Direction;
|
|
29
|
-
value: (candidate: T) => number;
|
|
30
|
-
}
|
|
31
|
-
interface ParetoResult<T> {
|
|
32
|
-
frontier: T[];
|
|
33
|
-
dominated: T[];
|
|
34
|
-
/** Index map: frontier[i] dominates each of dominatedBy[i]. */
|
|
35
|
-
dominanceMap: Array<{
|
|
36
|
-
dominator: T;
|
|
37
|
-
dominated: T[];
|
|
38
|
-
}>;
|
|
39
|
-
}
|
|
40
|
-
/** Does candidate A weakly dominate B — strictly better on at least one objective and no worse on any? */
|
|
41
|
-
declare function dominates<T>(a: T, b: T, objectives: Objective<T>[]): boolean;
|
|
42
|
-
/**
|
|
43
|
-
* Compute the non-dominated frontier. Candidates with NaN/Infinity on any
|
|
44
|
-
* objective are excluded (can't rank them). A candidate enters the frontier
|
|
45
|
-
* iff no other candidate dominates it.
|
|
46
|
-
*/
|
|
47
|
-
declare function paretoFrontier<T>(candidates: T[], objectives: Objective<T>[]): ParetoResult<T>;
|
|
48
|
-
/**
|
|
49
|
-
* Weighted-sum scalarisation. Use as a tie-break / single-winner selector
|
|
50
|
-
* when callers don't want to consume a frontier. Each objective contributes
|
|
51
|
-
* its normalised value (0..1 via min-max across the candidate pool) times
|
|
52
|
-
* its weight; missing weights default to 1/N.
|
|
53
|
-
*
|
|
54
|
-
* Direction is honoured automatically — `minimize` axes have their values
|
|
55
|
-
* inverted before scaling so "higher scalar = better" always holds.
|
|
56
|
-
*/
|
|
57
|
-
declare function scalarScore<T>(candidates: T[], objectives: Objective<T>[], options?: {
|
|
58
|
-
weights?: Partial<Record<string, number>>;
|
|
59
|
-
}): Array<{
|
|
60
|
-
candidate: T;
|
|
61
|
-
score: number;
|
|
62
|
-
}>;
|
|
63
|
-
/**
|
|
64
|
-
* NSGA-II crowding distance — secondary sort for ties on the frontier.
|
|
65
|
-
*
|
|
66
|
-
* When the Pareto front collapses to a single point (or many candidates tie
|
|
67
|
-
* on dominance), naive selection picks arbitrarily and the population
|
|
68
|
-
* degenerates over generations. NSGA-II preserves diversity by preferring
|
|
69
|
-
* candidates with more empty space around them on the frontier.
|
|
70
|
-
*
|
|
71
|
-
* Returns an array of `{ candidate, distance }` in the SAME order as the
|
|
72
|
-
* input. Higher distance = more isolated = should be preferred when
|
|
73
|
-
* preserving diversity.
|
|
74
|
-
*/
|
|
75
|
-
declare function crowdingDistance<T>(candidates: T[], objectives: Objective<T>[]): Array<{
|
|
76
|
-
candidate: T;
|
|
77
|
-
distance: number;
|
|
78
|
-
}>;
|
|
79
|
-
/**
|
|
80
|
-
* Pareto frontier with tie-break by crowding distance — the canonical
|
|
81
|
-
* NSGA-II selection step. Returns the frontier sorted by descending crowding
|
|
82
|
-
* distance so callers can `.slice(0, k)` to pick K diverse winners.
|
|
83
|
-
*/
|
|
84
|
-
declare function paretoFrontierWithCrowding<T>(candidates: T[], objectives: Objective<T>[]): Array<{
|
|
85
|
-
candidate: T;
|
|
86
|
-
distance: number;
|
|
87
|
-
}>;
|
|
88
|
-
|
|
89
|
-
/**
|
|
90
|
-
* Red-team battery — adversarial scenario corpus with per-category
|
|
91
|
-
* scorers.
|
|
92
|
-
*
|
|
93
|
-
* Every scenario is a `DatasetScenario` with a `redTeam: {...}` payload
|
|
94
|
-
* that declares: category, expected refusal, attack surface. Scorers
|
|
95
|
-
* inspect the agent's output + trajectory and return a PASS/FAIL per
|
|
96
|
-
* category. Aggregated via `redTeamReport`.
|
|
97
|
-
*
|
|
98
|
-
* The built-in corpus is intentionally conservative — 40 cases
|
|
99
|
-
* covering the most-seen attack patterns. Extend via `extendCorpus`.
|
|
100
|
-
*/
|
|
101
|
-
|
|
102
|
-
type RedTeamCategory = 'prompt_injection_direct' | 'prompt_injection_indirect' | 'jailbreak_persona' | 'jailbreak_dan' | 'pii_leak' | 'permission_escalation' | 'data_exfil' | 'policy_override';
|
|
103
|
-
interface RedTeamPayload {
|
|
104
|
-
category: RedTeamCategory;
|
|
105
|
-
/** The hostile input the agent sees. */
|
|
106
|
-
input: string;
|
|
107
|
-
/**
|
|
108
|
-
* What a safe agent should do: "refuse" means explicit refusal,
|
|
109
|
-
* "ignore" means process benignly without executing the attack.
|
|
110
|
-
*/
|
|
111
|
-
expected: 'refuse' | 'ignore';
|
|
112
|
-
/** Substring(s) that MUST NOT appear in output. */
|
|
113
|
-
forbiddenStrings?: string[];
|
|
114
|
-
/** Tool names the agent MUST NOT call for this scenario. */
|
|
115
|
-
forbiddenTools?: string[];
|
|
116
|
-
}
|
|
117
|
-
interface RedTeamCase extends DatasetScenario {
|
|
118
|
-
payload: RedTeamPayload;
|
|
119
|
-
}
|
|
120
|
-
interface RedTeamFinding {
|
|
121
|
-
scenarioId: string;
|
|
122
|
-
category: RedTeamCategory;
|
|
123
|
-
passed: boolean;
|
|
124
|
-
reason: string;
|
|
125
|
-
evidence?: string;
|
|
126
|
-
}
|
|
127
|
-
interface RedTeamReport {
|
|
128
|
-
findings: RedTeamFinding[];
|
|
129
|
-
passRateByCategory: Record<RedTeamCategory, number>;
|
|
130
|
-
overallPassRate: number;
|
|
131
|
-
}
|
|
132
|
-
/** Ship a canonical, small corpus. Consumers extend via `extendCorpus`. */
|
|
133
|
-
declare const DEFAULT_RED_TEAM_CORPUS: RedTeamCase[];
|
|
134
|
-
declare function redTeamDataset(extraCases?: RedTeamCase[]): Dataset;
|
|
135
|
-
/**
|
|
136
|
-
* Score a single agent response against the red-team case.
|
|
137
|
-
* `toolCalls` should be the names of tools the agent invoked during the run.
|
|
138
|
-
*/
|
|
139
|
-
declare function scoreRedTeamOutput(output: string, toolCalls: string[], rtCase: RedTeamCase): RedTeamFinding;
|
|
140
|
-
/** Aggregate red-team findings into per-category pass rates. */
|
|
141
|
-
declare function redTeamReport(findings: RedTeamFinding[]): RedTeamReport;
|
|
142
|
-
/**
|
|
143
|
-
* Extract the tool-call names from a corpus run — convenience for the
|
|
144
|
-
* common pipeline (run the scenario → score the run).
|
|
145
|
-
*/
|
|
146
|
-
declare function toolNamesForRun(store: TraceStore, runId: string): Promise<string[]>;
|
|
147
|
-
|
|
148
|
-
declare const REFERENCE_EQUIVALENCE_JUDGE_VERSION = "reference-equivalence-judge-v1-2026-07-13";
|
|
149
|
-
declare const REFERENCE_EQUIVALENCE_INPUT_LIMITS: {
|
|
150
|
-
readonly userRequest: 8000;
|
|
151
|
-
readonly expectedAnswer: 32000;
|
|
152
|
-
readonly candidateOutput: 32000;
|
|
153
|
-
};
|
|
154
|
-
interface ReferenceEquivalenceScenario extends Scenario {
|
|
155
|
-
userRequest: string;
|
|
156
|
-
expectedAnswer: string;
|
|
157
|
-
}
|
|
158
|
-
interface ReferenceEquivalenceJudgeInput {
|
|
159
|
-
userRequest: string;
|
|
160
|
-
expectedAnswer: string;
|
|
161
|
-
candidateOutput: string;
|
|
162
|
-
}
|
|
163
|
-
interface ReferenceEquivalenceJudgeOptions {
|
|
164
|
-
/** Injected transport. No implicit provider or credentials are selected. */
|
|
165
|
-
chat: ChatClient;
|
|
166
|
-
/** Falls back to the ChatClient's default model. */
|
|
167
|
-
model?: string;
|
|
168
|
-
/** Used only by the direct-call adapter. */
|
|
169
|
-
signal?: AbortSignal;
|
|
170
|
-
/** Optional receipt destination for direct calls; campaigns supply their own. */
|
|
171
|
-
costLedger?: CostLedger;
|
|
172
|
-
}
|
|
173
|
-
interface ReferenceEquivalenceJudgeResult extends LlmCallMetadata {
|
|
174
|
-
kind: 'reference-equivalence';
|
|
175
|
-
version: string;
|
|
176
|
-
score: number;
|
|
177
|
-
rationale: string;
|
|
178
|
-
}
|
|
179
|
-
/** Build the campaign-native expected-answer judge. */
|
|
180
|
-
declare function createReferenceEquivalenceJudge(options: ReferenceEquivalenceJudgeOptions): JudgeConfig<string, ReferenceEquivalenceScenario>;
|
|
181
|
-
/** Direct-call adapter over the campaign judge for product callers. */
|
|
182
|
-
declare function runReferenceEquivalenceJudge(input: ReferenceEquivalenceJudgeInput, options: ReferenceEquivalenceJudgeOptions): Promise<ReferenceEquivalenceJudgeResult>;
|
|
183
|
-
|
|
184
|
-
/**
|
|
185
|
-
* `openAutoPr` — thin shell-out helper for the `runImprovementLoop` preset's
|
|
186
|
-
* `autoOnPromote: 'pr'` mode. Substitutes for the per-product PR-opening
|
|
187
|
-
* code consumers duplicated 4 times. The PR body includes the campaign's
|
|
188
|
-
* manifest hash, gate verdict, and scorecard summary so reviewers can see
|
|
189
|
-
* exactly what was promoted + why.
|
|
190
|
-
*
|
|
191
|
-
* NOT a deploy mechanism — this only OPENS a PR. The human reviews + merges.
|
|
192
|
-
* The Shape B (`autoOnPromote: 'config'`) live-runtime-mutation path is
|
|
193
|
-
* deferred to Pass B with the full shadow / canary / rollback stack.
|
|
194
|
-
*/
|
|
195
|
-
|
|
196
|
-
interface OpenAutoPrOptions<TArtifact, TScenario extends Scenario> {
|
|
197
|
-
/** Campaign result to attach to the PR. */
|
|
198
|
-
result: CampaignResult<TArtifact, TScenario>;
|
|
199
|
-
/** Gate verdict explaining the promotion. Substrate refuses to open a PR
|
|
200
|
-
* when `gate.decision !== 'ship'` — fails loud. */
|
|
201
|
-
gate: GateResult;
|
|
202
|
-
/** Promoted surface diff — typically the new system prompt addendum or
|
|
203
|
-
* full profile diff. Substrate writes it as the PR body. */
|
|
204
|
-
promotedDiff: string;
|
|
205
|
-
/** GH owner/repo target (e.g., `tangle-network/gtm-agent`). */
|
|
206
|
-
ghOwner: string;
|
|
207
|
-
ghRepo: string;
|
|
208
|
-
/** Branch name for the PR. Default `auto/<manifestHash[:12]>`. */
|
|
209
|
-
branch?: string;
|
|
210
|
-
/** PR title. Default includes manifest hash. */
|
|
211
|
-
title?: string;
|
|
212
|
-
/** Whether to actually open the PR or just dry-run. Default reads
|
|
213
|
-
* `GH_AUTO_PR_TOKEN` env — present = open, absent = dry-run. */
|
|
214
|
-
dryRun?: boolean;
|
|
215
|
-
/** Test seam — substitute `gh pr create` invocation. */
|
|
216
|
-
ghExec?: (args: string[]) => {
|
|
217
|
-
stdout: string;
|
|
218
|
-
stderr: string;
|
|
219
|
-
status: number;
|
|
220
|
-
};
|
|
221
|
-
}
|
|
222
|
-
interface OpenAutoPrResult {
|
|
223
|
-
opened: boolean;
|
|
224
|
-
prUrl?: string;
|
|
225
|
-
dryRun: boolean;
|
|
226
|
-
reason: string;
|
|
227
|
-
}
|
|
228
|
-
/**
|
|
229
|
-
* Open a GitHub PR for a gate-approved surface promotion, attaching the manifest hash, gate verdict, and diff as the PR body.
|
|
230
|
-
*/
|
|
231
|
-
declare function openAutoPr<TArtifact, TScenario extends Scenario>(options: OpenAutoPrOptions<TArtifact, TScenario>): OpenAutoPrResult;
|
|
232
|
-
|
|
233
|
-
/**
|
|
234
|
-
* `runCampaign` — Pass A substrate primitive. ONE function that orchestrates
|
|
235
|
-
* scenarios → dispatch → artifacts → judges → aggregates, with full
|
|
236
|
-
* reproducibility (seed + manifest hash), cell-level resumability, bootstrap
|
|
237
|
-
* CIs, and the `LabeledScenarioStore` capture flywheel.
|
|
238
|
-
*
|
|
239
|
-
* Improvement loops (optimizer / gate / autoOnPromote) ride on top of this
|
|
240
|
-
* primitive but live in `presets/run-improvement-loop.ts`. This file keeps
|
|
241
|
-
* the core orchestrator minimal — Phase 1 of the Pass A track.
|
|
242
|
-
*/
|
|
243
|
-
|
|
244
|
-
interface RunCampaignOptions<TScenario extends Scenario, TArtifact> {
|
|
245
|
-
scenarios: TScenario[];
|
|
246
|
-
dispatch: DispatchFn<TScenario, TArtifact>;
|
|
247
|
-
/**
|
|
248
|
-
* Stable identity for the dispatch behavior, included in the manifest/cache
|
|
249
|
-
* key. Set this when the same function name can run different models,
|
|
250
|
-
* prompts, tools, or external config.
|
|
251
|
-
*/
|
|
252
|
-
dispatchRef?: string;
|
|
253
|
-
judges?: JudgeConfig<TArtifact, TScenario>[];
|
|
254
|
-
/** Required for reproducibility. Default 42. */
|
|
255
|
-
seed?: number;
|
|
256
|
-
/** Per-scenario replicates for CI bands. Default 1; raise to 5+ for
|
|
257
|
-
* bootstrap-tight intervals on critical eval. */
|
|
258
|
-
reps?: number;
|
|
259
|
-
/** When true (default), completed cells are cached by
|
|
260
|
-
* (manifestHash, scenarioId, rep, generation). Re-runs skip cached cells. */
|
|
261
|
-
resumable?: boolean;
|
|
262
|
-
/** Optional store — when present, every artifact + judge score is captured
|
|
263
|
-
* with the configured `captureSource`. Capture is default ON; pass `'off'`
|
|
264
|
-
* to disable. */
|
|
265
|
-
labeledStore?: LabeledScenarioStore | 'off';
|
|
266
|
-
captureSource?: 'production-trace' | 'eval-run' | 'manual' | 'red-team' | 'synthetic';
|
|
267
|
-
captureSourceVersionHash?: string;
|
|
268
|
-
/** Hard spend cap. Each paid call reserves its enforced maximum before dispatch. */
|
|
269
|
-
costCeiling?: number;
|
|
270
|
-
/** Shared spend account. Improvement loops pass one ledger through every
|
|
271
|
-
* campaign so the ceiling and returned total are run-wide. */
|
|
272
|
-
costLedger?: CostLedger;
|
|
273
|
-
/** Attribution label for receipts recorded by this campaign. */
|
|
274
|
-
costPhase?: string;
|
|
275
|
-
/** Max concurrent cells. Default 2. */
|
|
276
|
-
maxConcurrency?: number;
|
|
277
|
-
/**
|
|
278
|
-
* Per-cell dispatch deadline in ms. A `dispatch` that neither resolves nor
|
|
279
|
-
* rejects within this window is a hang (a stalled model request, an
|
|
280
|
-
* exhausted runtime resource, a backend that never closes its stream). When
|
|
281
|
-
* set, the cell's `ctx.signal` is aborted and the cell is recorded as a LOUD
|
|
282
|
-
* error (`dispatch exceeded <N>ms`) so the campaign proceeds and the failure
|
|
283
|
-
* is visible — instead of one wedged cell silently hanging the whole run (and
|
|
284
|
-
* every loop/CI job above it) forever. `undefined`/`0` = unbounded (legacy).
|
|
285
|
-
*/
|
|
286
|
-
dispatchTimeoutMs?: number;
|
|
287
|
-
/** Required: where artifacts + traces land. A bare name (not an absolute path)
|
|
288
|
-
* resolves to the shared `~/.tangle/traces/<repo>/runs/<name>` root so run
|
|
289
|
-
* bundles never pollute a repo working tree. Pass an absolute path to override. */
|
|
290
|
-
runDir: string;
|
|
291
|
-
/** Subject repo for the shared run-dir root (defaults to the CWD basename).
|
|
292
|
-
* Only consulted when `runDir` is a bare name. */
|
|
293
|
-
repo?: string;
|
|
294
|
-
/** Tracing posture. Default is the substrate's `FileSystemTraceStore` rooted
|
|
295
|
-
* at `<runDir>/traces/`. `'off'` disables capture entirely — substrate
|
|
296
|
-
* refuses this when the caller wires `autoOnPromote !== 'none'`. */
|
|
297
|
-
tracing?: 'on' | 'off';
|
|
298
|
-
/**
|
|
299
|
-
* Per-cell usage expectation — the early, fine-grained sibling of the
|
|
300
|
-
* batch `assertRealBackend` guard. A cell that produced an artifact (no
|
|
301
|
-
* error) but reported `costUsd === 0` AND zero tokens is a stub: the
|
|
302
|
-
* dispatch never reported LLM activity via `ctx.cost`. Modes:
|
|
303
|
-
* - `'warn'` (default) — log the offending cell loudly, keep going.
|
|
304
|
-
* - `'assert'` — throw `BackendIntegrityError` on the first such cell
|
|
305
|
-
* (fail-fast; recommended for CI campaigns expecting real LLM calls).
|
|
306
|
-
* - `'off'` — no check (replay / deterministic-only / offline analysis).
|
|
307
|
-
*/
|
|
308
|
-
expectUsage?: 'assert' | 'warn' | 'off';
|
|
309
|
-
/** Test seam — override the wall clock for deterministic tests. */
|
|
310
|
-
now?: () => Date;
|
|
311
|
-
/** Test seam — override per-cell trace writer factory. */
|
|
312
|
-
buildTraceWriter?: (cellId: string, dir: string) => CampaignTraceWriter;
|
|
313
|
-
/** Storage backend for run/cell dirs, the resumability cache, artifacts,
|
|
314
|
-
* and trace spans. Default: the Node filesystem (`fsCampaignStorage`).
|
|
315
|
-
* Pass `inMemoryCampaignStorage()` to run in a filesystem-less runtime
|
|
316
|
-
* (Cloudflare Workers, Deno, edge) — the `CampaignResult` is still
|
|
317
|
-
* produced; artifacts/traces just aren't persisted to disk. */
|
|
318
|
-
storage?: CampaignStorage;
|
|
319
|
-
/**
|
|
320
|
-
* Optional per-cell placement strategy. Returns an opaque string the
|
|
321
|
-
* substrate forwards as `ctx.placement` to the Dispatch — placement-aware
|
|
322
|
-
* Dispatches (e.g. `httpDispatch` from `/adapters/http`) use it to route
|
|
323
|
-
* each cell to the right worker, region, or sandbox. When unset, every
|
|
324
|
-
* cell receives `ctx.placement = undefined` and behaves identically to
|
|
325
|
-
* the in-process case.
|
|
326
|
-
*
|
|
327
|
-
* @example
|
|
328
|
-
* cellPlacement: ({ scenario }) => scenario.tags?.includes('eu') ? 'eu-west' : 'us-east'
|
|
329
|
-
*/
|
|
330
|
-
cellPlacement?: (input: {
|
|
331
|
-
scenario: TScenario;
|
|
332
|
-
rep: number;
|
|
333
|
-
generation?: number;
|
|
334
|
-
}) => string | undefined;
|
|
335
|
-
}
|
|
336
|
-
/**
|
|
337
|
-
* Core campaign orchestrator: fan scenarios through dispatch, score with judges, aggregate bootstrap CIs, and persist reproducible `CampaignResult` records.
|
|
338
|
-
*/
|
|
339
|
-
declare function runCampaign<TScenario extends Scenario, TArtifact>(opts: RunCampaignOptions<TScenario, TArtifact>): Promise<CampaignResult<TArtifact, TScenario>>;
|
|
340
|
-
interface CampaignRunPlanCell {
|
|
341
|
-
cellId: string;
|
|
342
|
-
scenarioId: string;
|
|
343
|
-
rep: number;
|
|
344
|
-
seed: number;
|
|
345
|
-
cachePath: string;
|
|
346
|
-
status: 'cached' | 'run';
|
|
347
|
-
reason?: 'missing' | 'manifest-mismatch' | 'cell-mismatch' | 'corrupt' | 'resumable-off';
|
|
348
|
-
}
|
|
349
|
-
interface CampaignRunPlan {
|
|
350
|
-
manifestHash: string;
|
|
351
|
-
totalCells: number;
|
|
352
|
-
cellsCached: number;
|
|
353
|
-
cellsToRun: number;
|
|
354
|
-
cells: CampaignRunPlanCell[];
|
|
355
|
-
}
|
|
356
|
-
interface PlanCampaignRunOptions<TScenario extends Scenario, TArtifact> {
|
|
357
|
-
scenarios: TScenario[];
|
|
358
|
-
dispatch?: DispatchFn<TScenario, TArtifact>;
|
|
359
|
-
dispatchRef?: string;
|
|
360
|
-
judges?: JudgeConfig<TArtifact, TScenario>[];
|
|
361
|
-
seed?: number;
|
|
362
|
-
reps?: number;
|
|
363
|
-
resumable?: boolean;
|
|
364
|
-
runDir: string;
|
|
365
|
-
/** Subject repo for the shared run-dir root (see RunCampaignOptions.repo). */
|
|
366
|
-
repo?: string;
|
|
367
|
-
storage?: CampaignStorage;
|
|
368
|
-
}
|
|
369
|
-
/**
|
|
370
|
-
* Plan a campaign WITHOUT dispatching: computes the manifest hash and the per-cell
|
|
371
|
-
* run-vs-cached schedule so callers can preview cost and resumability before spending.
|
|
372
|
-
*/
|
|
373
|
-
declare function planCampaignRun<TScenario extends Scenario, TArtifact>(opts: PlanCampaignRunOptions<TScenario, TArtifact>): CampaignRunPlan;
|
|
374
|
-
|
|
375
|
-
/**
|
|
376
|
-
* `runOptimization` — the improvement loop body. Runs N generations: the
|
|
377
|
-
* `SurfaceProposer` proposes K candidate surfaces per generation, each
|
|
378
|
-
* candidate runs a campaign (the measurement), and only a candidate that beats
|
|
379
|
-
* the single global incumbent becomes the next generation's parent.
|
|
380
|
-
* Proposer-agnostic — the same loop runs an evolutionary population mutator
|
|
381
|
-
* (`evolutionaryProposer`) or any reflective / agentic proposer; they differ
|
|
382
|
-
* only in how `propose()` picks candidates.
|
|
383
|
-
*
|
|
384
|
-
* This is `runLoop`'s shape (plan → measure → decide) specialized to surface
|
|
385
|
-
* improvement: `proposer.propose` = plan, `runCampaign` = the measurement
|
|
386
|
-
* (which runs the worker behind `dispatch`), the mean-composite ranking = the
|
|
387
|
-
* validator, `proposer.decide` = the stop check.
|
|
388
|
-
*
|
|
389
|
-
* The gated-promotion shell (`runImprovementLoop`) wraps this with a holdout
|
|
390
|
-
* re-score + release gate + optional PR.
|
|
391
|
-
*/
|
|
392
|
-
|
|
393
|
-
interface RunOptimizationBaseOptions<TScenario extends Scenario, TArtifact> extends Omit<RunCampaignOptions<TScenario, TArtifact>, 'dispatch'> {
|
|
394
|
-
/** Initial mutable surface (typically system prompt or addendum). */
|
|
395
|
-
baselineSurface: MutableSurface;
|
|
396
|
-
/** Dispatcher that takes the CURRENT surface + scenario → artifact. */
|
|
397
|
-
dispatchWithSurface: (surface: MutableSurface, scenario: TScenario, ctx: Parameters<RunCampaignOptions<TScenario, TArtifact>['dispatch']>[1]) => Promise<TArtifact>;
|
|
398
|
-
/** The candidate-generation strategy. Wrap a population `Mutator` via
|
|
399
|
-
* `evolutionaryProposer({ mutator })`, or pass any reflective / agentic
|
|
400
|
-
* proposer that implements `SurfaceProposer`. */
|
|
401
|
-
proposer: SurfaceProposer;
|
|
402
|
-
populationSize: number;
|
|
403
|
-
maxGenerations: number;
|
|
404
|
-
/** @deprecated The loop has one global incumbent and can promote only the
|
|
405
|
-
* single candidate that beats it. Retained for source compatibility. */
|
|
406
|
-
promoteTopK?: number;
|
|
407
|
-
/** DEPTH knob forwarded to the proposer's `propose()` — max iterations the
|
|
408
|
-
* agentic generator may take per candidate. */
|
|
409
|
-
maxImprovementShots?: number;
|
|
410
|
-
/** Optional analysis report forwarded to `propose()`. Opaque here; the
|
|
411
|
-
* proposer types it. */
|
|
412
|
-
report?: unknown;
|
|
413
|
-
/** Structured findings forwarded to `propose()` as `ctx.findings`. A
|
|
414
|
-
* findings producer emits these from the
|
|
415
|
-
* generation's traces; findings-grounded proposers consume them. Opaque here;
|
|
416
|
-
* the proposer types its `TFindings`. Empty when no producer is wired. */
|
|
417
|
-
findings?: unknown[];
|
|
418
|
-
/** Per-generation findings producer. Runs once on the BASELINE campaign
|
|
419
|
-
* (as `generation: -1`, the baseline convention) before generation 0
|
|
420
|
-
* proposes — so even a single-generation run proposes with trace context —
|
|
421
|
-
* and then after each generation's candidates are scored with that
|
|
422
|
-
* generation's results; whatever it returns REPLACES `ctx.findings` for the
|
|
423
|
-
* NEXT `propose()`, so the diagnosis is refreshed each round instead
|
|
424
|
-
* of being a static one-shot. Generic by design: the substrate does not
|
|
425
|
-
* import an analyst — the consumer plugs its trace-analyst registry / HALO
|
|
426
|
-
* here (reading the per-candidate `runDir` traces). When absent, findings
|
|
427
|
-
* stay the static `opts.findings`. */
|
|
428
|
-
analyzeGeneration?: (input: {
|
|
429
|
-
generation: number;
|
|
430
|
-
runDir: string;
|
|
431
|
-
candidates: Array<{
|
|
432
|
-
surfaceHash: string;
|
|
433
|
-
campaign: CampaignResult<TArtifact, TScenario>;
|
|
434
|
-
composite: number;
|
|
435
|
-
}>;
|
|
436
|
-
history: GenerationRecord[];
|
|
437
|
-
/** Shared run spend account and receipt attribution phase. */
|
|
438
|
-
costLedger?: CostLedger;
|
|
439
|
-
costPhase?: string;
|
|
440
|
-
}) => Promise<unknown[]>;
|
|
441
|
-
}
|
|
442
|
-
type RunOptimizationOptions<TScenario extends Scenario, TArtifact> = RunOptimizationBaseOptions<TScenario, TArtifact>;
|
|
443
|
-
interface RunOptimizationResult<TArtifact, TScenario extends Scenario> {
|
|
444
|
-
generations: Array<{
|
|
445
|
-
record: GenerationRecord;
|
|
446
|
-
surfaces: Array<{
|
|
447
|
-
surfaceHash: string;
|
|
448
|
-
surface: MutableSurface;
|
|
449
|
-
campaign: CampaignResult<TArtifact, TScenario>;
|
|
450
|
-
}>;
|
|
451
|
-
}>;
|
|
452
|
-
winnerSurface: MutableSurface;
|
|
453
|
-
winnerSurfaceHash: string;
|
|
454
|
-
/** Proposer label for the promoted surface. Present when the winning
|
|
455
|
-
* candidate came from a `ProposedCandidate` (a reflective proposer);
|
|
456
|
-
* absent when the winner is the baseline or a bare-surface mutator. */
|
|
457
|
-
winnerLabel?: string;
|
|
458
|
-
/** Proposer rationale for the promoted surface — the "because Z" that
|
|
459
|
-
* motivated the winning change. Survives to `SelfImproveResult` and the
|
|
460
|
-
* emitted provenance record. Absent when the winner is the baseline. */
|
|
461
|
-
winnerRationale?: string;
|
|
462
|
-
baselineCampaign: CampaignResult<TArtifact, TScenario>;
|
|
463
|
-
/** Run-wide spend, including agents, proposers, analysts, and judges. */
|
|
464
|
-
cost: CostLedgerSummary;
|
|
465
|
-
/** The GEPA Pareto frontier across every scored surface (baseline + all
|
|
466
|
-
* generations) by per-scenario objective vector — the non-dominated set.
|
|
467
|
-
* Each generation's `propose()` received the frontier-so-far as
|
|
468
|
-
* `ctx.paretoParents`; this is the final frontier. A surface here that is
|
|
469
|
-
* NOT the winner is uniquely best on some scenario the winner loses on. */
|
|
470
|
-
paretoFrontier: ParetoParent[];
|
|
471
|
-
}
|
|
472
|
-
/**
|
|
473
|
-
* Improvement loop body: N generations of propose → campaign → rank, maintaining a Pareto frontier and one global incumbent across generations.
|
|
474
|
-
*/
|
|
475
|
-
declare function runOptimization<TScenario extends Scenario, TArtifact>(opts: RunOptimizationOptions<TScenario, TArtifact>): Promise<RunOptimizationResult<TArtifact, TScenario>>;
|
|
476
|
-
|
|
477
|
-
/**
|
|
478
|
-
* `runImprovementLoop` — the gated-promotion shell around the improvement
|
|
479
|
-
* loop body (`runOptimization`). Proposes candidate surfaces via the
|
|
480
|
-
* `SurfaceProposer`, re-scores the winner against the baseline on a
|
|
481
|
-
* holdout set, runs the release gate, and optionally opens a PR.
|
|
482
|
-
*
|
|
483
|
-
* Role vocabulary (see docs/design/loop-taxonomy.md):
|
|
484
|
-
* - PROPOSER = the `SurfaceProposer` (evolutionary GEPA mutator OR
|
|
485
|
-
* reflective analyst). Proposes candidate SURFACES — the
|
|
486
|
-
* worker's system prompt / tool config — NOT conversation
|
|
487
|
-
* turns.
|
|
488
|
-
* - MEASUREMENT= `runCampaign`. Scores one surface by running the worker
|
|
489
|
-
* (via `dispatch`) over scenarios and judging the output.
|
|
490
|
-
* - WORKER = the agent harness in the sandbox, invoked behind the
|
|
491
|
-
* topology-opaque `dispatch` seam — never referenced here.
|
|
492
|
-
*
|
|
493
|
-
* Distinct from `runLoop` in `@tangle-network/agent-runtime`, which is the
|
|
494
|
-
* INNER conversation loop (execution driver ↔ workers in a sandbox). `runImprovementLoop`
|
|
495
|
-
* is the OUTER loop: it improves the surface that those workers run.
|
|
496
|
-
*
|
|
497
|
-
* Hard-refuses unsafe configurations:
|
|
498
|
-
* - `tracing: 'off'` when a proposer is wired (improvement is unattributable)
|
|
499
|
-
* - `autoOnPromote: 'config'` — DEFERRED to Pass B; v0.40 only ships
|
|
500
|
-
* `'pr'` and `'none'`.
|
|
501
|
-
*/
|
|
502
|
-
|
|
503
|
-
type RunImprovementLoopOptions<TScenario extends Scenario, TArtifact> = RunOptimizationOptions<TScenario, TArtifact> & {
|
|
504
|
-
/** Holdout scenarios kept OUT of the training optimization pool — used
|
|
505
|
-
* ONLY to score baseline vs winner for the gate. */
|
|
506
|
-
holdoutScenarios: TScenario[];
|
|
507
|
-
/** Promotion gate. Substrate strongly recommends `defaultProductionGate`
|
|
508
|
-
* for production wiring (composes red-team / reward-hacking / canary /
|
|
509
|
-
* heldout). */
|
|
510
|
-
gate: Gate<TArtifact, TScenario>;
|
|
511
|
-
/** What to do when the gate ships:
|
|
512
|
-
* - `'pr'`: open a PR via `openAutoPr`
|
|
513
|
-
* - `'none'`: just report — caller decides what to do with the winner
|
|
514
|
-
* v0.40 does NOT support `'config'` (live-runtime self-mutation) —
|
|
515
|
-
* deferred to Pass B behind safety stack. */
|
|
516
|
-
autoOnPromote: 'pr' | 'none';
|
|
517
|
-
/** GH owner / repo for the auto-PR. Required when autoOnPromote === 'pr'. */
|
|
518
|
-
ghOwner?: string;
|
|
519
|
-
ghRepo?: string;
|
|
520
|
-
/** Optional render override — substrate writes a diff-shaped surface; pass
|
|
521
|
-
* a function to format the promoted surface differently. */
|
|
522
|
-
renderPromotedDiff?: (winnerSurface: MutableSurface, baselineSurface: MutableSurface) => string;
|
|
523
|
-
/** Placebo control. When supplied AND the winner differs from baseline, the
|
|
524
|
-
* loop scores a THIRD holdout arm: the winner surface with its content
|
|
525
|
-
* footprint-matched-blanked by this function (typically via `neutralizeText`).
|
|
526
|
-
* Its scores are exposed to the gate as `ctx.neutralizedJudgeScores`, letting
|
|
527
|
-
* a `neutralizationGate` reject a win whose lift survives blanking the content
|
|
528
|
-
* (decorative — driven by footprint, not content). Costs one extra holdout
|
|
529
|
-
* campaign; omit to skip. Return a byte/layout-matched blank of the winner. */
|
|
530
|
-
neutralize?: (winnerSurface: MutableSurface, baselineSurface: MutableSurface) => MutableSurface;
|
|
531
|
-
};
|
|
532
|
-
interface RunImprovementLoopResult<TArtifact, TScenario extends Scenario> extends RunOptimizationResult<TArtifact, TScenario> {
|
|
533
|
-
baselineOnHoldout: CampaignResult<TArtifact, TScenario>;
|
|
534
|
-
winnerOnHoldout: CampaignResult<TArtifact, TScenario>;
|
|
535
|
-
gateResult: Awaited<ReturnType<Gate<TArtifact, TScenario>['decide']>>;
|
|
536
|
-
/** Unified baseline→winner surface diff. Computed UNCONDITIONALLY (not only
|
|
537
|
-
* when `autoOnPromote === 'pr'`) so the diff that the gate decided on is
|
|
538
|
-
* always present on the result + in the emitted provenance record. Empty
|
|
539
|
-
* string when winner == baseline (no change to diff). */
|
|
540
|
-
promotedDiff: string;
|
|
541
|
-
prResult?: ReturnType<typeof openAutoPr>;
|
|
542
|
-
}
|
|
543
|
-
/**
|
|
544
|
-
* Gated-promotion shell over `runOptimization`: scores the winner against the baseline on a holdout set, runs the release gate, and optionally opens a PR.
|
|
545
|
-
*/
|
|
546
|
-
declare function runImprovementLoop<TScenario extends Scenario, TArtifact>(opts: RunImprovementLoopOptions<TScenario, TArtifact>): Promise<RunImprovementLoopResult<TArtifact, TScenario>>;
|
|
547
|
-
/**
|
|
548
|
-
* Default surface diff renderer: produces a unified baseline/winner text diff for prompt surfaces or a worktree-ref summary for code surfaces.
|
|
549
|
-
*/
|
|
550
|
-
declare function defaultRenderDiff(winnerSurface: MutableSurface, baselineSurface: MutableSurface): string;
|
|
551
|
-
|
|
552
|
-
/**
|
|
553
|
-
* `gepaProposer` — a reflective `SurfaceProposer` for prompt-tier surfaces.
|
|
554
|
-
* Each generation it reflects on the prior best candidate's per-scenario
|
|
555
|
-
* scores + weakest dimensions, asks an LLM to propose targeted rewrites of
|
|
556
|
-
* the current surface, and returns them as the next population.
|
|
557
|
-
*
|
|
558
|
-
* Maps onto the GEPA paper (Agrawal et al., arXiv:2507.19457):
|
|
559
|
-
* - *Reflection*: each generation reflects on the best parent's weakest
|
|
560
|
-
* dimensions + per-scenario top/bottom scores to propose targeted rewrites.
|
|
561
|
-
* - *Pareto frontier*: `runOptimization` maintains the non-dominated set of
|
|
562
|
-
* surfaces across generations (per-scenario objective vectors) and supplies
|
|
563
|
-
* it as `ctx.paretoParents`. A surface uniquely best on one hard scenario
|
|
564
|
-
* survives even when its mean composite is lower.
|
|
565
|
-
* - *Combine complementary lessons*: when the frontier has >1 member, the
|
|
566
|
-
* first population slot is a merge of those parents' strengths (one LLM
|
|
567
|
-
* call citing each parent's winning scenarios). Toggle via `combineParents`.
|
|
568
|
-
* Dominance is computed by the package-canonical `paretoFrontier` (`pareto.ts`).
|
|
569
|
-
*
|
|
570
|
-
* Optional `constraints` move structured-doc guards into the proposer
|
|
571
|
-
* (preserve H2 section headings, cap sentence-level edits) — useful when
|
|
572
|
-
* the surface IS a structured procedure like a SKILL.md / runbook /
|
|
573
|
-
* judge rubric. When `constraints` is omitted, behavior is unchanged.
|
|
574
|
-
*
|
|
575
|
-
* The proposer is surface-agnostic — any string surface in any consumer opts
|
|
576
|
-
* in by selecting it. Reuses the generic reflection primitive
|
|
577
|
-
* (`buildReflectionPrompt` / `parseReflectionResponse`) and the router client.
|
|
578
|
-
*
|
|
579
|
-
* Earns its keep where there is real per-instance signal (which the
|
|
580
|
-
* dimensional + per-scenario evidence + the `LabeledScenarioStore` flywheel
|
|
581
|
-
* now provide). For thin-signal surfaces it degrades to plain reflection.
|
|
582
|
-
* On generation 0 (no history) it reflects on the current surface against
|
|
583
|
-
* the mutation primitives alone.
|
|
584
|
-
*/
|
|
585
|
-
|
|
586
|
-
interface GepaProposerConstraints {
|
|
587
|
-
/** H2 section headings that MUST appear unchanged in every candidate.
|
|
588
|
-
* When set, the proposer auto-detects current H2s if this is empty AND
|
|
589
|
-
* rejects any candidate that drops or renames a preserved heading.
|
|
590
|
-
* Use when the surface is a structured doc (SKILL.md, runbook,
|
|
591
|
-
* sectioned system prompt, judge rubric). */
|
|
592
|
-
preserveSections?: string[];
|
|
593
|
-
/** Maximum sentence-level edits per candidate vs the parent surface.
|
|
594
|
-
* Rejection threshold = maxSentenceEdits × 2 (counts adds + removes).
|
|
595
|
-
* Inspired by SkillOpt's edit-budget as a "textual learning rate."
|
|
596
|
-
* Cap prevents an LLM rewrite from overwriting useful prior rules. */
|
|
597
|
-
maxSentenceEdits?: number;
|
|
598
|
-
}
|
|
599
|
-
interface GepaProposerOptions {
|
|
600
|
-
/** Router transport (apiKey/baseUrl). */
|
|
601
|
-
llm: LlmClientOptions;
|
|
602
|
-
/** Model that performs the reflection. */
|
|
603
|
-
model: string;
|
|
604
|
-
/** Optional ledger for direct proposer use. Campaign context takes precedence. */
|
|
605
|
-
costLedger?: CostLedger;
|
|
606
|
-
/** What is being optimized — appears in the reflection prompt for orientation. */
|
|
607
|
-
target: string;
|
|
608
|
-
/** Surface-specific mutation levers offered to the model. */
|
|
609
|
-
mutationPrimitives?: string[];
|
|
610
|
-
/** Top/bottom scenarios surfaced as evidence each generation. Default 3. */
|
|
611
|
-
evidenceK?: number;
|
|
612
|
-
/** Reflection sampling temperature. Default 0.7. */
|
|
613
|
-
temperature?: number;
|
|
614
|
-
/** Reflection max tokens. Default 6000. */
|
|
615
|
-
maxTokens?: number;
|
|
616
|
-
/** Structured-doc constraints. Candidates violating any are rejected
|
|
617
|
-
* post-parse and dropped from the returned population. */
|
|
618
|
-
constraints?: GepaProposerConstraints;
|
|
619
|
-
/** GEPA combine-complementary-lessons: when the loop supplies a Pareto
|
|
620
|
-
* frontier of >1 non-dominated parents (`ctx.paretoParents`), spend one
|
|
621
|
-
* slot of the population on a merge of their strengths. Default `true` —
|
|
622
|
-
* this is the GEPA-faithful behavior; the merge only fires once the
|
|
623
|
-
* frontier has more than one member (generation ≥ 1). Set `false` for
|
|
624
|
-
* pure single-parent reflection. */
|
|
625
|
-
combineParents?: boolean;
|
|
626
|
-
/** Cap on how many frontier parents feed one combine prompt (highest
|
|
627
|
-
* composite first), to bound prompt size. Default 4. */
|
|
628
|
-
combineMaxParents?: number;
|
|
629
|
-
}
|
|
630
|
-
/**
|
|
631
|
-
* GEPA reflective proposer: each generation reflects on the weakest scenarios and dimensions to produce targeted prompt rewrites, optionally combining Pareto-frontier parents.
|
|
632
|
-
*/
|
|
633
|
-
declare function gepaProposer(opts: GepaProposerOptions): SurfaceProposer;
|
|
634
|
-
/** Extract H2 headings (`## Foo`) from a markdown surface. Exported for
|
|
635
|
-
* consumers building custom mutators that share the same invariant. */
|
|
636
|
-
declare function extractH2Sections(text: string): string[];
|
|
637
|
-
/** Sentence-level edit distance — count distinct add/remove ops between
|
|
638
|
-
* two surfaces via a normalised line-by-line set diff. Treats trivial
|
|
639
|
-
* whitespace as identical. Exported for tests + consumer-side validators. */
|
|
640
|
-
declare function countSentenceEdits(baseline: string, candidate: string): number;
|
|
641
|
-
|
|
642
|
-
export { type ParetoResult as A, DEFAULT_RED_TEAM_CORPUS as B, type CampaignRunPlan as C, type Direction as D, type RedTeamCategory as E, type RedTeamFinding as F, type GepaProposerOptions as G, type RedTeamPayload as H, type RedTeamReport as I, crowdingDistance as J, dominates as K, paretoFrontier as L, paretoFrontierWithCrowding as M, redTeamDataset as N, type OpenAutoPrOptions as O, type PlanCampaignRunOptions as P, redTeamReport as Q, type RunOptimizationOptions as R, scalarScore as S, scoreRedTeamOutput as T, toolNamesForRun as U, type RunImprovementLoopResult as a, REFERENCE_EQUIVALENCE_INPUT_LIMITS as b, REFERENCE_EQUIVALENCE_JUDGE_VERSION as c, type ReferenceEquivalenceJudgeInput as d, type ReferenceEquivalenceJudgeOptions as e, type ReferenceEquivalenceJudgeResult as f, type ReferenceEquivalenceScenario as g, type RunCampaignOptions as h, type RunImprovementLoopOptions as i, createReferenceEquivalenceJudge as j, gepaProposer as k, runImprovementLoop as l, runReferenceEquivalenceJudge as m, type RedTeamCase as n, type CampaignRunPlanCell as o, type GepaProposerConstraints as p, type OpenAutoPrResult as q, runCampaign as r, type RunOptimizationResult as s, countSentenceEdits as t, defaultRenderDiff as u, extractH2Sections as v, openAutoPr as w, planCampaignRun as x, runOptimization as y, type Objective as z };
|