@tangle-network/agent-eval 0.117.1 → 0.118.1

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (143) hide show
  1. package/CHANGELOG.md +30 -0
  2. package/dist/analyst/index.d.ts +2772 -21
  3. package/dist/analyst/index.js +7 -6
  4. package/dist/analyst/index.js.map +1 -1
  5. package/dist/belief-state/index.d.ts +706 -10
  6. package/dist/belief-state/index.js +2 -1
  7. package/dist/belief-state/index.js.map +1 -1
  8. package/dist/benchmarks/index.d.ts +958 -14
  9. package/dist/benchmarks/index.js +12 -10
  10. package/dist/builder-eval/index.d.ts +449 -4
  11. package/dist/builder-eval/index.js +4 -3
  12. package/dist/builder-eval/index.js.map +1 -1
  13. package/dist/campaign/index.d.ts +4275 -73
  14. package/dist/campaign/index.js +12 -10
  15. package/dist/{chunk-VF3XSYTI.js → chunk-33JA4TFA.js} +6 -6
  16. package/dist/{chunk-4JLWXDYA.js → chunk-3EHHMC6E.js} +2 -2
  17. package/dist/{chunk-CCZIVI3F.js → chunk-BFW56GTT.js} +2 -2
  18. package/dist/{chunk-YZPO4UHR.js → chunk-FTUMG2U7.js} +124 -149
  19. package/dist/chunk-FTUMG2U7.js.map +1 -0
  20. package/dist/{chunk-E4BUPP7Z.js → chunk-HKUCJ437.js} +38 -66
  21. package/dist/chunk-HKUCJ437.js.map +1 -0
  22. package/dist/{chunk-HZHNRYHK.js → chunk-K6N6XJJX.js} +2 -2
  23. package/dist/chunk-KSDQVPLR.js +286 -0
  24. package/dist/chunk-KSDQVPLR.js.map +1 -0
  25. package/dist/chunk-MA6HLL3S.js +65 -0
  26. package/dist/chunk-MA6HLL3S.js.map +1 -0
  27. package/dist/{chunk-MGEHEHSN.js → chunk-OIIMMLRB.js} +11 -11
  28. package/dist/{chunk-DXZRATT5.js → chunk-OYZAPX5G.js} +3 -3
  29. package/dist/chunk-PXE2VKMX.js +140 -0
  30. package/dist/chunk-PXE2VKMX.js.map +1 -0
  31. package/dist/{chunk-JSJZ4PJ6.js → chunk-Q442S5AS.js} +17 -17
  32. package/dist/{chunk-ODVOOEWQ.js → chunk-QBRSJK47.js} +2 -2
  33. package/dist/{chunk-S2F4J57L.js → chunk-QKEGNI5B.js} +77 -32
  34. package/dist/chunk-QKEGNI5B.js.map +1 -0
  35. package/dist/{chunk-5UF54T55.js → chunk-S3UZOQ5Y.js} +34 -7
  36. package/dist/chunk-S3UZOQ5Y.js.map +1 -0
  37. package/dist/{chunk-FQNLDL4D.js → chunk-SVH2ANFD.js} +136 -4
  38. package/dist/chunk-SVH2ANFD.js.map +1 -0
  39. package/dist/{chunk-GQCZRZ7L.js → chunk-U5CHZ5M3.js} +9 -9
  40. package/dist/{chunk-TVVP3ZZQ.js → chunk-VQMK5FMP.js} +2 -1
  41. package/dist/{chunk-TVVP3ZZQ.js.map → chunk-VQMK5FMP.js.map} +1 -1
  42. package/dist/chunk-WDHBCA3M.js +31 -0
  43. package/dist/chunk-WDHBCA3M.js.map +1 -0
  44. package/dist/{chunk-HQPHZGL6.js → chunk-YLMUS4MM.js} +9 -9
  45. package/dist/{chunk-LQUTGLOZ.js → chunk-ZET2UAYW.js} +15 -65
  46. package/dist/chunk-ZET2UAYW.js.map +1 -0
  47. package/dist/contract/index.d.ts +4012 -38
  48. package/dist/contract/index.js +56 -29
  49. package/dist/contract/index.js.map +1 -1
  50. package/dist/control.d.ts +1013 -9
  51. package/dist/control.js +4 -3
  52. package/dist/fuzz.d.ts +194 -4
  53. package/dist/fuzz.js +3 -3
  54. package/dist/hosted/index.d.ts +498 -17
  55. package/dist/index.d.ts +10992 -1288
  56. package/dist/index.js +101 -80
  57. package/dist/index.js.map +1 -1
  58. package/dist/matrix/index.d.ts +139 -4
  59. package/dist/meta-eval/index.d.ts +862 -15
  60. package/dist/meta-eval/index.js +2 -1
  61. package/dist/meta-eval/index.js.map +1 -1
  62. package/dist/multishot/index.d.ts +214 -14
  63. package/dist/openapi.json +1 -1
  64. package/dist/pipelines/index.d.ts +392 -7
  65. package/dist/pipelines/index.js +5 -3
  66. package/dist/pipelines/index.js.map +1 -1
  67. package/dist/reporting.d.ts +1277 -17
  68. package/dist/rl.d.ts +2359 -28
  69. package/dist/rl.js +6 -5
  70. package/dist/rl.js.map +1 -1
  71. package/dist/storyboard/index.d.ts +86 -1
  72. package/dist/trace-attributes.d.ts +16 -0
  73. package/dist/trace-attributes.js +32 -0
  74. package/dist/trace-attributes.js.map +1 -0
  75. package/dist/traces.d.ts +1978 -697
  76. package/dist/traces.js +53 -32
  77. package/dist/wire/index.d.ts +655 -9
  78. package/docs/insight-report.md +44 -0
  79. package/package.json +9 -3
  80. package/dist/adversarial-B7loGVVX.d.ts +0 -19
  81. package/dist/analyst-C8HHvfJp.d.ts +0 -88
  82. package/dist/analyze-runs--2x39HZ7.d.ts +0 -81
  83. package/dist/baseline-DKq3gJpP.d.ts +0 -141
  84. package/dist/calibration-C8MTS7cw.d.ts +0 -101
  85. package/dist/chunk-5UF54T55.js.map +0 -1
  86. package/dist/chunk-E4BUPP7Z.js.map +0 -1
  87. package/dist/chunk-FQNLDL4D.js.map +0 -1
  88. package/dist/chunk-LQUTGLOZ.js.map +0 -1
  89. package/dist/chunk-S2F4J57L.js.map +0 -1
  90. package/dist/chunk-YZPO4UHR.js.map +0 -1
  91. package/dist/code-agent-session-CjZsVd19.d.ts +0 -87
  92. package/dist/control-6vuGfmDH.d.ts +0 -258
  93. package/dist/cost-ledger-DWy3XdJc.d.ts +0 -183
  94. package/dist/dataset-NENEzRgk.d.ts +0 -115
  95. package/dist/default-registry-DaK8b3fv.d.ts +0 -155
  96. package/dist/emitter-CjD7vUwv.d.ts +0 -122
  97. package/dist/errors-oeQrLqXC.d.ts +0 -74
  98. package/dist/failure-cluster-DOAcSJ87.d.ts +0 -76
  99. package/dist/feedback-trajectory-BUnM58xL.d.ts +0 -348
  100. package/dist/gepa-eESocoDi.d.ts +0 -642
  101. package/dist/index-PdX4VnPA.d.ts +0 -423
  102. package/dist/insight-report-DY4nDW9Q.d.ts +0 -310
  103. package/dist/integrity-DqlBiLyK.d.ts +0 -81
  104. package/dist/judge-calibration-7C-IDmKr.d.ts +0 -145
  105. package/dist/kind-factory-ClZmO25A.d.ts +0 -171
  106. package/dist/llm-client-qoDd18Qz.d.ts +0 -289
  107. package/dist/multi-layer-verifier-BsqKuLyN.d.ts +0 -150
  108. package/dist/off-policy-DiwuKKg7.d.ts +0 -132
  109. package/dist/outcome-store-rnXLEqSn.d.ts +0 -63
  110. package/dist/policy-edit-wG9uFEFm.d.ts +0 -455
  111. package/dist/pre-registration-BWQhJ3vz.d.ts +0 -761
  112. package/dist/provenance-DpjwyseI.d.ts +0 -541
  113. package/dist/query-CF7PG61p.d.ts +0 -35
  114. package/dist/raw-provider-sink-C46HDghv.d.ts +0 -132
  115. package/dist/release-report-C8G2i5Xi.d.ts +0 -236
  116. package/dist/researcher-C8XyxQsu.d.ts +0 -387
  117. package/dist/rubric-predictive-validity-p49lLVrE.d.ts +0 -105
  118. package/dist/run-record-BDH49H2E.d.ts +0 -360
  119. package/dist/runtime-trajectory-DGBIUt4B.d.ts +0 -49
  120. package/dist/schema-B3Q3l9Z_.d.ts +0 -201
  121. package/dist/semantic-concept-judge-CXnPEJbf.d.ts +0 -723
  122. package/dist/sequential-5iSVfzl2.d.ts +0 -139
  123. package/dist/series-convergence-D5OWMBg6.d.ts +0 -33
  124. package/dist/statistics-KUnG73jH.d.ts +0 -494
  125. package/dist/storage-DrX3v_5B.d.ts +0 -50
  126. package/dist/store-C1YxJDEK.d.ts +0 -248
  127. package/dist/store-DGqD0Pyo.d.ts +0 -116
  128. package/dist/summary-report-C5bKFfm-.d.ts +0 -445
  129. package/dist/test-graded-scenario-B0ybnPY7.d.ts +0 -166
  130. package/dist/types-BSw1rOUB.d.ts +0 -634
  131. package/dist/types-BUxNaJ8c.d.ts +0 -108
  132. package/dist/types-BkfcQnxV.d.ts +0 -313
  133. package/dist/verdict-C9MlYujm.d.ts +0 -35
  134. /package/dist/{chunk-VF3XSYTI.js.map → chunk-33JA4TFA.js.map} +0 -0
  135. /package/dist/{chunk-4JLWXDYA.js.map → chunk-3EHHMC6E.js.map} +0 -0
  136. /package/dist/{chunk-CCZIVI3F.js.map → chunk-BFW56GTT.js.map} +0 -0
  137. /package/dist/{chunk-HZHNRYHK.js.map → chunk-K6N6XJJX.js.map} +0 -0
  138. /package/dist/{chunk-MGEHEHSN.js.map → chunk-OIIMMLRB.js.map} +0 -0
  139. /package/dist/{chunk-DXZRATT5.js.map → chunk-OYZAPX5G.js.map} +0 -0
  140. /package/dist/{chunk-JSJZ4PJ6.js.map → chunk-Q442S5AS.js.map} +0 -0
  141. /package/dist/{chunk-ODVOOEWQ.js.map → chunk-QBRSJK47.js.map} +0 -0
  142. /package/dist/{chunk-GQCZRZ7L.js.map → chunk-U5CHZ5M3.js.map} +0 -0
  143. /package/dist/{chunk-HQPHZGL6.js.map → chunk-YLMUS4MM.js.map} +0 -0
@@ -1,642 +0,0 @@
1
- import { D as DatasetScenario, c as Dataset } from './dataset-NENEzRgk.js';
2
- import { T as TraceStore } from './store-DGqD0Pyo.js';
3
- import { C as ChatClient } from './policy-edit-wG9uFEFm.js';
4
- import { S as Scenario, b as JudgeConfig, C as CampaignResult, m as GateResult, k as DispatchFn, L as LabeledScenarioStore, i as CampaignTraceWriter, o as GenerationRecord, M as MutableSurface, P as ParetoParent, c as SurfaceProposer, G as Gate } from './types-BSw1rOUB.js';
5
- import { C as CostLedger, b as CostLedgerSummary } from './cost-ledger-DWy3XdJc.js';
6
- import { L as LlmCallMetadata, a as LlmClientOptions } from './llm-client-qoDd18Qz.js';
7
- import { C as CampaignStorage } from './storage-DrX3v_5B.js';
8
-
9
- /**
10
- * Pareto frontier — multi-objective optimization over candidate runs.
11
- *
12
- * Lifted from ADC pareto.ts and blueprint-agent frontier.ts. When you're
13
- * trading off (cost, latency, quality) or (passRate, tokenBudget,
14
- * ttfb), you rarely have a single "winner" — you have a set of
15
- * non-dominated candidates. This module exposes:
16
- *
17
- * - `paretoFrontier`: filter a set of candidates to the non-dominated ones
18
- * - `dominates`: does A dominate B across all objectives?
19
- *
20
- * Each objective is declared with a direction: 'maximize' (higher=better)
21
- * or 'minimize' (lower=better). Candidates are any object; pass an
22
- * `objective(candidate)` accessor.
23
- */
24
- type Direction = 'maximize' | 'minimize';
25
- interface Objective<T> {
26
- /** Stable label used in reports. */
27
- name: string;
28
- direction: Direction;
29
- value: (candidate: T) => number;
30
- }
31
- interface ParetoResult<T> {
32
- frontier: T[];
33
- dominated: T[];
34
- /** Index map: frontier[i] dominates each of dominatedBy[i]. */
35
- dominanceMap: Array<{
36
- dominator: T;
37
- dominated: T[];
38
- }>;
39
- }
40
- /** Does candidate A weakly dominate B — strictly better on at least one objective and no worse on any? */
41
- declare function dominates<T>(a: T, b: T, objectives: Objective<T>[]): boolean;
42
- /**
43
- * Compute the non-dominated frontier. Candidates with NaN/Infinity on any
44
- * objective are excluded (can't rank them). A candidate enters the frontier
45
- * iff no other candidate dominates it.
46
- */
47
- declare function paretoFrontier<T>(candidates: T[], objectives: Objective<T>[]): ParetoResult<T>;
48
- /**
49
- * Weighted-sum scalarisation. Use as a tie-break / single-winner selector
50
- * when callers don't want to consume a frontier. Each objective contributes
51
- * its normalised value (0..1 via min-max across the candidate pool) times
52
- * its weight; missing weights default to 1/N.
53
- *
54
- * Direction is honoured automatically — `minimize` axes have their values
55
- * inverted before scaling so "higher scalar = better" always holds.
56
- */
57
- declare function scalarScore<T>(candidates: T[], objectives: Objective<T>[], options?: {
58
- weights?: Partial<Record<string, number>>;
59
- }): Array<{
60
- candidate: T;
61
- score: number;
62
- }>;
63
- /**
64
- * NSGA-II crowding distance — secondary sort for ties on the frontier.
65
- *
66
- * When the Pareto front collapses to a single point (or many candidates tie
67
- * on dominance), naive selection picks arbitrarily and the population
68
- * degenerates over generations. NSGA-II preserves diversity by preferring
69
- * candidates with more empty space around them on the frontier.
70
- *
71
- * Returns an array of `{ candidate, distance }` in the SAME order as the
72
- * input. Higher distance = more isolated = should be preferred when
73
- * preserving diversity.
74
- */
75
- declare function crowdingDistance<T>(candidates: T[], objectives: Objective<T>[]): Array<{
76
- candidate: T;
77
- distance: number;
78
- }>;
79
- /**
80
- * Pareto frontier with tie-break by crowding distance — the canonical
81
- * NSGA-II selection step. Returns the frontier sorted by descending crowding
82
- * distance so callers can `.slice(0, k)` to pick K diverse winners.
83
- */
84
- declare function paretoFrontierWithCrowding<T>(candidates: T[], objectives: Objective<T>[]): Array<{
85
- candidate: T;
86
- distance: number;
87
- }>;
88
-
89
- /**
90
- * Red-team battery — adversarial scenario corpus with per-category
91
- * scorers.
92
- *
93
- * Every scenario is a `DatasetScenario` with a `redTeam: {...}` payload
94
- * that declares: category, expected refusal, attack surface. Scorers
95
- * inspect the agent's output + trajectory and return a PASS/FAIL per
96
- * category. Aggregated via `redTeamReport`.
97
- *
98
- * The built-in corpus is intentionally conservative — 40 cases
99
- * covering the most-seen attack patterns. Extend via `extendCorpus`.
100
- */
101
-
102
- type RedTeamCategory = 'prompt_injection_direct' | 'prompt_injection_indirect' | 'jailbreak_persona' | 'jailbreak_dan' | 'pii_leak' | 'permission_escalation' | 'data_exfil' | 'policy_override';
103
- interface RedTeamPayload {
104
- category: RedTeamCategory;
105
- /** The hostile input the agent sees. */
106
- input: string;
107
- /**
108
- * What a safe agent should do: "refuse" means explicit refusal,
109
- * "ignore" means process benignly without executing the attack.
110
- */
111
- expected: 'refuse' | 'ignore';
112
- /** Substring(s) that MUST NOT appear in output. */
113
- forbiddenStrings?: string[];
114
- /** Tool names the agent MUST NOT call for this scenario. */
115
- forbiddenTools?: string[];
116
- }
117
- interface RedTeamCase extends DatasetScenario {
118
- payload: RedTeamPayload;
119
- }
120
- interface RedTeamFinding {
121
- scenarioId: string;
122
- category: RedTeamCategory;
123
- passed: boolean;
124
- reason: string;
125
- evidence?: string;
126
- }
127
- interface RedTeamReport {
128
- findings: RedTeamFinding[];
129
- passRateByCategory: Record<RedTeamCategory, number>;
130
- overallPassRate: number;
131
- }
132
- /** Ship a canonical, small corpus. Consumers extend via `extendCorpus`. */
133
- declare const DEFAULT_RED_TEAM_CORPUS: RedTeamCase[];
134
- declare function redTeamDataset(extraCases?: RedTeamCase[]): Dataset;
135
- /**
136
- * Score a single agent response against the red-team case.
137
- * `toolCalls` should be the names of tools the agent invoked during the run.
138
- */
139
- declare function scoreRedTeamOutput(output: string, toolCalls: string[], rtCase: RedTeamCase): RedTeamFinding;
140
- /** Aggregate red-team findings into per-category pass rates. */
141
- declare function redTeamReport(findings: RedTeamFinding[]): RedTeamReport;
142
- /**
143
- * Extract the tool-call names from a corpus run — convenience for the
144
- * common pipeline (run the scenario → score the run).
145
- */
146
- declare function toolNamesForRun(store: TraceStore, runId: string): Promise<string[]>;
147
-
148
- declare const REFERENCE_EQUIVALENCE_JUDGE_VERSION = "reference-equivalence-judge-v1-2026-07-13";
149
- declare const REFERENCE_EQUIVALENCE_INPUT_LIMITS: {
150
- readonly userRequest: 8000;
151
- readonly expectedAnswer: 32000;
152
- readonly candidateOutput: 32000;
153
- };
154
- interface ReferenceEquivalenceScenario extends Scenario {
155
- userRequest: string;
156
- expectedAnswer: string;
157
- }
158
- interface ReferenceEquivalenceJudgeInput {
159
- userRequest: string;
160
- expectedAnswer: string;
161
- candidateOutput: string;
162
- }
163
- interface ReferenceEquivalenceJudgeOptions {
164
- /** Injected transport. No implicit provider or credentials are selected. */
165
- chat: ChatClient;
166
- /** Falls back to the ChatClient's default model. */
167
- model?: string;
168
- /** Used only by the direct-call adapter. */
169
- signal?: AbortSignal;
170
- /** Optional receipt destination for direct calls; campaigns supply their own. */
171
- costLedger?: CostLedger;
172
- }
173
- interface ReferenceEquivalenceJudgeResult extends LlmCallMetadata {
174
- kind: 'reference-equivalence';
175
- version: string;
176
- score: number;
177
- rationale: string;
178
- }
179
- /** Build the campaign-native expected-answer judge. */
180
- declare function createReferenceEquivalenceJudge(options: ReferenceEquivalenceJudgeOptions): JudgeConfig<string, ReferenceEquivalenceScenario>;
181
- /** Direct-call adapter over the campaign judge for product callers. */
182
- declare function runReferenceEquivalenceJudge(input: ReferenceEquivalenceJudgeInput, options: ReferenceEquivalenceJudgeOptions): Promise<ReferenceEquivalenceJudgeResult>;
183
-
184
- /**
185
- * `openAutoPr` — thin shell-out helper for the `runImprovementLoop` preset's
186
- * `autoOnPromote: 'pr'` mode. Substitutes for the per-product PR-opening
187
- * code consumers duplicated 4 times. The PR body includes the campaign's
188
- * manifest hash, gate verdict, and scorecard summary so reviewers can see
189
- * exactly what was promoted + why.
190
- *
191
- * NOT a deploy mechanism — this only OPENS a PR. The human reviews + merges.
192
- * The Shape B (`autoOnPromote: 'config'`) live-runtime-mutation path is
193
- * deferred to Pass B with the full shadow / canary / rollback stack.
194
- */
195
-
196
- interface OpenAutoPrOptions<TArtifact, TScenario extends Scenario> {
197
- /** Campaign result to attach to the PR. */
198
- result: CampaignResult<TArtifact, TScenario>;
199
- /** Gate verdict explaining the promotion. Substrate refuses to open a PR
200
- * when `gate.decision !== 'ship'` — fails loud. */
201
- gate: GateResult;
202
- /** Promoted surface diff — typically the new system prompt addendum or
203
- * full profile diff. Substrate writes it as the PR body. */
204
- promotedDiff: string;
205
- /** GH owner/repo target (e.g., `tangle-network/gtm-agent`). */
206
- ghOwner: string;
207
- ghRepo: string;
208
- /** Branch name for the PR. Default `auto/<manifestHash[:12]>`. */
209
- branch?: string;
210
- /** PR title. Default includes manifest hash. */
211
- title?: string;
212
- /** Whether to actually open the PR or just dry-run. Default reads
213
- * `GH_AUTO_PR_TOKEN` env — present = open, absent = dry-run. */
214
- dryRun?: boolean;
215
- /** Test seam — substitute `gh pr create` invocation. */
216
- ghExec?: (args: string[]) => {
217
- stdout: string;
218
- stderr: string;
219
- status: number;
220
- };
221
- }
222
- interface OpenAutoPrResult {
223
- opened: boolean;
224
- prUrl?: string;
225
- dryRun: boolean;
226
- reason: string;
227
- }
228
- /**
229
- * Open a GitHub PR for a gate-approved surface promotion, attaching the manifest hash, gate verdict, and diff as the PR body.
230
- */
231
- declare function openAutoPr<TArtifact, TScenario extends Scenario>(options: OpenAutoPrOptions<TArtifact, TScenario>): OpenAutoPrResult;
232
-
233
- /**
234
- * `runCampaign` — Pass A substrate primitive. ONE function that orchestrates
235
- * scenarios → dispatch → artifacts → judges → aggregates, with full
236
- * reproducibility (seed + manifest hash), cell-level resumability, bootstrap
237
- * CIs, and the `LabeledScenarioStore` capture flywheel.
238
- *
239
- * Improvement loops (optimizer / gate / autoOnPromote) ride on top of this
240
- * primitive but live in `presets/run-improvement-loop.ts`. This file keeps
241
- * the core orchestrator minimal — Phase 1 of the Pass A track.
242
- */
243
-
244
- interface RunCampaignOptions<TScenario extends Scenario, TArtifact> {
245
- scenarios: TScenario[];
246
- dispatch: DispatchFn<TScenario, TArtifact>;
247
- /**
248
- * Stable identity for the dispatch behavior, included in the manifest/cache
249
- * key. Set this when the same function name can run different models,
250
- * prompts, tools, or external config.
251
- */
252
- dispatchRef?: string;
253
- judges?: JudgeConfig<TArtifact, TScenario>[];
254
- /** Required for reproducibility. Default 42. */
255
- seed?: number;
256
- /** Per-scenario replicates for CI bands. Default 1; raise to 5+ for
257
- * bootstrap-tight intervals on critical eval. */
258
- reps?: number;
259
- /** When true (default), completed cells are cached by
260
- * (manifestHash, scenarioId, rep, generation). Re-runs skip cached cells. */
261
- resumable?: boolean;
262
- /** Optional store — when present, every artifact + judge score is captured
263
- * with the configured `captureSource`. Capture is default ON; pass `'off'`
264
- * to disable. */
265
- labeledStore?: LabeledScenarioStore | 'off';
266
- captureSource?: 'production-trace' | 'eval-run' | 'manual' | 'red-team' | 'synthetic';
267
- captureSourceVersionHash?: string;
268
- /** Hard spend cap. Each paid call reserves its enforced maximum before dispatch. */
269
- costCeiling?: number;
270
- /** Shared spend account. Improvement loops pass one ledger through every
271
- * campaign so the ceiling and returned total are run-wide. */
272
- costLedger?: CostLedger;
273
- /** Attribution label for receipts recorded by this campaign. */
274
- costPhase?: string;
275
- /** Max concurrent cells. Default 2. */
276
- maxConcurrency?: number;
277
- /**
278
- * Per-cell dispatch deadline in ms. A `dispatch` that neither resolves nor
279
- * rejects within this window is a hang (a stalled model request, an
280
- * exhausted runtime resource, a backend that never closes its stream). When
281
- * set, the cell's `ctx.signal` is aborted and the cell is recorded as a LOUD
282
- * error (`dispatch exceeded <N>ms`) so the campaign proceeds and the failure
283
- * is visible — instead of one wedged cell silently hanging the whole run (and
284
- * every loop/CI job above it) forever. `undefined`/`0` = unbounded (legacy).
285
- */
286
- dispatchTimeoutMs?: number;
287
- /** Required: where artifacts + traces land. A bare name (not an absolute path)
288
- * resolves to the shared `~/.tangle/traces/<repo>/runs/<name>` root so run
289
- * bundles never pollute a repo working tree. Pass an absolute path to override. */
290
- runDir: string;
291
- /** Subject repo for the shared run-dir root (defaults to the CWD basename).
292
- * Only consulted when `runDir` is a bare name. */
293
- repo?: string;
294
- /** Tracing posture. Default is the substrate's `FileSystemTraceStore` rooted
295
- * at `<runDir>/traces/`. `'off'` disables capture entirely — substrate
296
- * refuses this when the caller wires `autoOnPromote !== 'none'`. */
297
- tracing?: 'on' | 'off';
298
- /**
299
- * Per-cell usage expectation — the early, fine-grained sibling of the
300
- * batch `assertRealBackend` guard. A cell that produced an artifact (no
301
- * error) but reported `costUsd === 0` AND zero tokens is a stub: the
302
- * dispatch never reported LLM activity via `ctx.cost`. Modes:
303
- * - `'warn'` (default) — log the offending cell loudly, keep going.
304
- * - `'assert'` — throw `BackendIntegrityError` on the first such cell
305
- * (fail-fast; recommended for CI campaigns expecting real LLM calls).
306
- * - `'off'` — no check (replay / deterministic-only / offline analysis).
307
- */
308
- expectUsage?: 'assert' | 'warn' | 'off';
309
- /** Test seam — override the wall clock for deterministic tests. */
310
- now?: () => Date;
311
- /** Test seam — override per-cell trace writer factory. */
312
- buildTraceWriter?: (cellId: string, dir: string) => CampaignTraceWriter;
313
- /** Storage backend for run/cell dirs, the resumability cache, artifacts,
314
- * and trace spans. Default: the Node filesystem (`fsCampaignStorage`).
315
- * Pass `inMemoryCampaignStorage()` to run in a filesystem-less runtime
316
- * (Cloudflare Workers, Deno, edge) — the `CampaignResult` is still
317
- * produced; artifacts/traces just aren't persisted to disk. */
318
- storage?: CampaignStorage;
319
- /**
320
- * Optional per-cell placement strategy. Returns an opaque string the
321
- * substrate forwards as `ctx.placement` to the Dispatch — placement-aware
322
- * Dispatches (e.g. `httpDispatch` from `/adapters/http`) use it to route
323
- * each cell to the right worker, region, or sandbox. When unset, every
324
- * cell receives `ctx.placement = undefined` and behaves identically to
325
- * the in-process case.
326
- *
327
- * @example
328
- * cellPlacement: ({ scenario }) => scenario.tags?.includes('eu') ? 'eu-west' : 'us-east'
329
- */
330
- cellPlacement?: (input: {
331
- scenario: TScenario;
332
- rep: number;
333
- generation?: number;
334
- }) => string | undefined;
335
- }
336
- /**
337
- * Core campaign orchestrator: fan scenarios through dispatch, score with judges, aggregate bootstrap CIs, and persist reproducible `CampaignResult` records.
338
- */
339
- declare function runCampaign<TScenario extends Scenario, TArtifact>(opts: RunCampaignOptions<TScenario, TArtifact>): Promise<CampaignResult<TArtifact, TScenario>>;
340
- interface CampaignRunPlanCell {
341
- cellId: string;
342
- scenarioId: string;
343
- rep: number;
344
- seed: number;
345
- cachePath: string;
346
- status: 'cached' | 'run';
347
- reason?: 'missing' | 'manifest-mismatch' | 'cell-mismatch' | 'corrupt' | 'resumable-off';
348
- }
349
- interface CampaignRunPlan {
350
- manifestHash: string;
351
- totalCells: number;
352
- cellsCached: number;
353
- cellsToRun: number;
354
- cells: CampaignRunPlanCell[];
355
- }
356
- interface PlanCampaignRunOptions<TScenario extends Scenario, TArtifact> {
357
- scenarios: TScenario[];
358
- dispatch?: DispatchFn<TScenario, TArtifact>;
359
- dispatchRef?: string;
360
- judges?: JudgeConfig<TArtifact, TScenario>[];
361
- seed?: number;
362
- reps?: number;
363
- resumable?: boolean;
364
- runDir: string;
365
- /** Subject repo for the shared run-dir root (see RunCampaignOptions.repo). */
366
- repo?: string;
367
- storage?: CampaignStorage;
368
- }
369
- /**
370
- * Plan a campaign WITHOUT dispatching: computes the manifest hash and the per-cell
371
- * run-vs-cached schedule so callers can preview cost and resumability before spending.
372
- */
373
- declare function planCampaignRun<TScenario extends Scenario, TArtifact>(opts: PlanCampaignRunOptions<TScenario, TArtifact>): CampaignRunPlan;
374
-
375
- /**
376
- * `runOptimization` — the improvement loop body. Runs N generations: the
377
- * `SurfaceProposer` proposes K candidate surfaces per generation, each
378
- * candidate runs a campaign (the measurement), and only a candidate that beats
379
- * the single global incumbent becomes the next generation's parent.
380
- * Proposer-agnostic — the same loop runs an evolutionary population mutator
381
- * (`evolutionaryProposer`) or any reflective / agentic proposer; they differ
382
- * only in how `propose()` picks candidates.
383
- *
384
- * This is `runLoop`'s shape (plan → measure → decide) specialized to surface
385
- * improvement: `proposer.propose` = plan, `runCampaign` = the measurement
386
- * (which runs the worker behind `dispatch`), the mean-composite ranking = the
387
- * validator, `proposer.decide` = the stop check.
388
- *
389
- * The gated-promotion shell (`runImprovementLoop`) wraps this with a holdout
390
- * re-score + release gate + optional PR.
391
- */
392
-
393
- interface RunOptimizationBaseOptions<TScenario extends Scenario, TArtifact> extends Omit<RunCampaignOptions<TScenario, TArtifact>, 'dispatch'> {
394
- /** Initial mutable surface (typically system prompt or addendum). */
395
- baselineSurface: MutableSurface;
396
- /** Dispatcher that takes the CURRENT surface + scenario → artifact. */
397
- dispatchWithSurface: (surface: MutableSurface, scenario: TScenario, ctx: Parameters<RunCampaignOptions<TScenario, TArtifact>['dispatch']>[1]) => Promise<TArtifact>;
398
- /** The candidate-generation strategy. Wrap a population `Mutator` via
399
- * `evolutionaryProposer({ mutator })`, or pass any reflective / agentic
400
- * proposer that implements `SurfaceProposer`. */
401
- proposer: SurfaceProposer;
402
- populationSize: number;
403
- maxGenerations: number;
404
- /** @deprecated The loop has one global incumbent and can promote only the
405
- * single candidate that beats it. Retained for source compatibility. */
406
- promoteTopK?: number;
407
- /** DEPTH knob forwarded to the proposer's `propose()` — max iterations the
408
- * agentic generator may take per candidate. */
409
- maxImprovementShots?: number;
410
- /** Optional analysis report forwarded to `propose()`. Opaque here; the
411
- * proposer types it. */
412
- report?: unknown;
413
- /** Structured findings forwarded to `propose()` as `ctx.findings`. A
414
- * findings producer emits these from the
415
- * generation's traces; findings-grounded proposers consume them. Opaque here;
416
- * the proposer types its `TFindings`. Empty when no producer is wired. */
417
- findings?: unknown[];
418
- /** Per-generation findings producer. Runs once on the BASELINE campaign
419
- * (as `generation: -1`, the baseline convention) before generation 0
420
- * proposes — so even a single-generation run proposes with trace context —
421
- * and then after each generation's candidates are scored with that
422
- * generation's results; whatever it returns REPLACES `ctx.findings` for the
423
- * NEXT `propose()`, so the diagnosis is refreshed each round instead
424
- * of being a static one-shot. Generic by design: the substrate does not
425
- * import an analyst — the consumer plugs its trace-analyst registry / HALO
426
- * here (reading the per-candidate `runDir` traces). When absent, findings
427
- * stay the static `opts.findings`. */
428
- analyzeGeneration?: (input: {
429
- generation: number;
430
- runDir: string;
431
- candidates: Array<{
432
- surfaceHash: string;
433
- campaign: CampaignResult<TArtifact, TScenario>;
434
- composite: number;
435
- }>;
436
- history: GenerationRecord[];
437
- /** Shared run spend account and receipt attribution phase. */
438
- costLedger?: CostLedger;
439
- costPhase?: string;
440
- }) => Promise<unknown[]>;
441
- }
442
- type RunOptimizationOptions<TScenario extends Scenario, TArtifact> = RunOptimizationBaseOptions<TScenario, TArtifact>;
443
- interface RunOptimizationResult<TArtifact, TScenario extends Scenario> {
444
- generations: Array<{
445
- record: GenerationRecord;
446
- surfaces: Array<{
447
- surfaceHash: string;
448
- surface: MutableSurface;
449
- campaign: CampaignResult<TArtifact, TScenario>;
450
- }>;
451
- }>;
452
- winnerSurface: MutableSurface;
453
- winnerSurfaceHash: string;
454
- /** Proposer label for the promoted surface. Present when the winning
455
- * candidate came from a `ProposedCandidate` (a reflective proposer);
456
- * absent when the winner is the baseline or a bare-surface mutator. */
457
- winnerLabel?: string;
458
- /** Proposer rationale for the promoted surface — the "because Z" that
459
- * motivated the winning change. Survives to `SelfImproveResult` and the
460
- * emitted provenance record. Absent when the winner is the baseline. */
461
- winnerRationale?: string;
462
- baselineCampaign: CampaignResult<TArtifact, TScenario>;
463
- /** Run-wide spend, including agents, proposers, analysts, and judges. */
464
- cost: CostLedgerSummary;
465
- /** The GEPA Pareto frontier across every scored surface (baseline + all
466
- * generations) by per-scenario objective vector — the non-dominated set.
467
- * Each generation's `propose()` received the frontier-so-far as
468
- * `ctx.paretoParents`; this is the final frontier. A surface here that is
469
- * NOT the winner is uniquely best on some scenario the winner loses on. */
470
- paretoFrontier: ParetoParent[];
471
- }
472
- /**
473
- * Improvement loop body: N generations of propose → campaign → rank, maintaining a Pareto frontier and one global incumbent across generations.
474
- */
475
- declare function runOptimization<TScenario extends Scenario, TArtifact>(opts: RunOptimizationOptions<TScenario, TArtifact>): Promise<RunOptimizationResult<TArtifact, TScenario>>;
476
-
477
- /**
478
- * `runImprovementLoop` — the gated-promotion shell around the improvement
479
- * loop body (`runOptimization`). Proposes candidate surfaces via the
480
- * `SurfaceProposer`, re-scores the winner against the baseline on a
481
- * holdout set, runs the release gate, and optionally opens a PR.
482
- *
483
- * Role vocabulary (see docs/design/loop-taxonomy.md):
484
- * - PROPOSER = the `SurfaceProposer` (evolutionary GEPA mutator OR
485
- * reflective analyst). Proposes candidate SURFACES — the
486
- * worker's system prompt / tool config — NOT conversation
487
- * turns.
488
- * - MEASUREMENT= `runCampaign`. Scores one surface by running the worker
489
- * (via `dispatch`) over scenarios and judging the output.
490
- * - WORKER = the agent harness in the sandbox, invoked behind the
491
- * topology-opaque `dispatch` seam — never referenced here.
492
- *
493
- * Distinct from `runLoop` in `@tangle-network/agent-runtime`, which is the
494
- * INNER conversation loop (execution driver ↔ workers in a sandbox). `runImprovementLoop`
495
- * is the OUTER loop: it improves the surface that those workers run.
496
- *
497
- * Hard-refuses unsafe configurations:
498
- * - `tracing: 'off'` when a proposer is wired (improvement is unattributable)
499
- * - `autoOnPromote: 'config'` — DEFERRED to Pass B; v0.40 only ships
500
- * `'pr'` and `'none'`.
501
- */
502
-
503
- type RunImprovementLoopOptions<TScenario extends Scenario, TArtifact> = RunOptimizationOptions<TScenario, TArtifact> & {
504
- /** Holdout scenarios kept OUT of the training optimization pool — used
505
- * ONLY to score baseline vs winner for the gate. */
506
- holdoutScenarios: TScenario[];
507
- /** Promotion gate. Substrate strongly recommends `defaultProductionGate`
508
- * for production wiring (composes red-team / reward-hacking / canary /
509
- * heldout). */
510
- gate: Gate<TArtifact, TScenario>;
511
- /** What to do when the gate ships:
512
- * - `'pr'`: open a PR via `openAutoPr`
513
- * - `'none'`: just report — caller decides what to do with the winner
514
- * v0.40 does NOT support `'config'` (live-runtime self-mutation) —
515
- * deferred to Pass B behind safety stack. */
516
- autoOnPromote: 'pr' | 'none';
517
- /** GH owner / repo for the auto-PR. Required when autoOnPromote === 'pr'. */
518
- ghOwner?: string;
519
- ghRepo?: string;
520
- /** Optional render override — substrate writes a diff-shaped surface; pass
521
- * a function to format the promoted surface differently. */
522
- renderPromotedDiff?: (winnerSurface: MutableSurface, baselineSurface: MutableSurface) => string;
523
- /** Placebo control. When supplied AND the winner differs from baseline, the
524
- * loop scores a THIRD holdout arm: the winner surface with its content
525
- * footprint-matched-blanked by this function (typically via `neutralizeText`).
526
- * Its scores are exposed to the gate as `ctx.neutralizedJudgeScores`, letting
527
- * a `neutralizationGate` reject a win whose lift survives blanking the content
528
- * (decorative — driven by footprint, not content). Costs one extra holdout
529
- * campaign; omit to skip. Return a byte/layout-matched blank of the winner. */
530
- neutralize?: (winnerSurface: MutableSurface, baselineSurface: MutableSurface) => MutableSurface;
531
- };
532
- interface RunImprovementLoopResult<TArtifact, TScenario extends Scenario> extends RunOptimizationResult<TArtifact, TScenario> {
533
- baselineOnHoldout: CampaignResult<TArtifact, TScenario>;
534
- winnerOnHoldout: CampaignResult<TArtifact, TScenario>;
535
- gateResult: Awaited<ReturnType<Gate<TArtifact, TScenario>['decide']>>;
536
- /** Unified baseline→winner surface diff. Computed UNCONDITIONALLY (not only
537
- * when `autoOnPromote === 'pr'`) so the diff that the gate decided on is
538
- * always present on the result + in the emitted provenance record. Empty
539
- * string when winner == baseline (no change to diff). */
540
- promotedDiff: string;
541
- prResult?: ReturnType<typeof openAutoPr>;
542
- }
543
- /**
544
- * Gated-promotion shell over `runOptimization`: scores the winner against the baseline on a holdout set, runs the release gate, and optionally opens a PR.
545
- */
546
- declare function runImprovementLoop<TScenario extends Scenario, TArtifact>(opts: RunImprovementLoopOptions<TScenario, TArtifact>): Promise<RunImprovementLoopResult<TArtifact, TScenario>>;
547
- /**
548
- * Default surface diff renderer: produces a unified baseline/winner text diff for prompt surfaces or a worktree-ref summary for code surfaces.
549
- */
550
- declare function defaultRenderDiff(winnerSurface: MutableSurface, baselineSurface: MutableSurface): string;
551
-
552
- /**
553
- * `gepaProposer` — a reflective `SurfaceProposer` for prompt-tier surfaces.
554
- * Each generation it reflects on the prior best candidate's per-scenario
555
- * scores + weakest dimensions, asks an LLM to propose targeted rewrites of
556
- * the current surface, and returns them as the next population.
557
- *
558
- * Maps onto the GEPA paper (Agrawal et al., arXiv:2507.19457):
559
- * - *Reflection*: each generation reflects on the best parent's weakest
560
- * dimensions + per-scenario top/bottom scores to propose targeted rewrites.
561
- * - *Pareto frontier*: `runOptimization` maintains the non-dominated set of
562
- * surfaces across generations (per-scenario objective vectors) and supplies
563
- * it as `ctx.paretoParents`. A surface uniquely best on one hard scenario
564
- * survives even when its mean composite is lower.
565
- * - *Combine complementary lessons*: when the frontier has >1 member, the
566
- * first population slot is a merge of those parents' strengths (one LLM
567
- * call citing each parent's winning scenarios). Toggle via `combineParents`.
568
- * Dominance is computed by the package-canonical `paretoFrontier` (`pareto.ts`).
569
- *
570
- * Optional `constraints` move structured-doc guards into the proposer
571
- * (preserve H2 section headings, cap sentence-level edits) — useful when
572
- * the surface IS a structured procedure like a SKILL.md / runbook /
573
- * judge rubric. When `constraints` is omitted, behavior is unchanged.
574
- *
575
- * The proposer is surface-agnostic — any string surface in any consumer opts
576
- * in by selecting it. Reuses the generic reflection primitive
577
- * (`buildReflectionPrompt` / `parseReflectionResponse`) and the router client.
578
- *
579
- * Earns its keep where there is real per-instance signal (which the
580
- * dimensional + per-scenario evidence + the `LabeledScenarioStore` flywheel
581
- * now provide). For thin-signal surfaces it degrades to plain reflection.
582
- * On generation 0 (no history) it reflects on the current surface against
583
- * the mutation primitives alone.
584
- */
585
-
586
- interface GepaProposerConstraints {
587
- /** H2 section headings that MUST appear unchanged in every candidate.
588
- * When set, the proposer auto-detects current H2s if this is empty AND
589
- * rejects any candidate that drops or renames a preserved heading.
590
- * Use when the surface is a structured doc (SKILL.md, runbook,
591
- * sectioned system prompt, judge rubric). */
592
- preserveSections?: string[];
593
- /** Maximum sentence-level edits per candidate vs the parent surface.
594
- * Rejection threshold = maxSentenceEdits × 2 (counts adds + removes).
595
- * Inspired by SkillOpt's edit-budget as a "textual learning rate."
596
- * Cap prevents an LLM rewrite from overwriting useful prior rules. */
597
- maxSentenceEdits?: number;
598
- }
599
- interface GepaProposerOptions {
600
- /** Router transport (apiKey/baseUrl). */
601
- llm: LlmClientOptions;
602
- /** Model that performs the reflection. */
603
- model: string;
604
- /** Optional ledger for direct proposer use. Campaign context takes precedence. */
605
- costLedger?: CostLedger;
606
- /** What is being optimized — appears in the reflection prompt for orientation. */
607
- target: string;
608
- /** Surface-specific mutation levers offered to the model. */
609
- mutationPrimitives?: string[];
610
- /** Top/bottom scenarios surfaced as evidence each generation. Default 3. */
611
- evidenceK?: number;
612
- /** Reflection sampling temperature. Default 0.7. */
613
- temperature?: number;
614
- /** Reflection max tokens. Default 6000. */
615
- maxTokens?: number;
616
- /** Structured-doc constraints. Candidates violating any are rejected
617
- * post-parse and dropped from the returned population. */
618
- constraints?: GepaProposerConstraints;
619
- /** GEPA combine-complementary-lessons: when the loop supplies a Pareto
620
- * frontier of >1 non-dominated parents (`ctx.paretoParents`), spend one
621
- * slot of the population on a merge of their strengths. Default `true` —
622
- * this is the GEPA-faithful behavior; the merge only fires once the
623
- * frontier has more than one member (generation ≥ 1). Set `false` for
624
- * pure single-parent reflection. */
625
- combineParents?: boolean;
626
- /** Cap on how many frontier parents feed one combine prompt (highest
627
- * composite first), to bound prompt size. Default 4. */
628
- combineMaxParents?: number;
629
- }
630
- /**
631
- * GEPA reflective proposer: each generation reflects on the weakest scenarios and dimensions to produce targeted prompt rewrites, optionally combining Pareto-frontier parents.
632
- */
633
- declare function gepaProposer(opts: GepaProposerOptions): SurfaceProposer;
634
- /** Extract H2 headings (`## Foo`) from a markdown surface. Exported for
635
- * consumers building custom mutators that share the same invariant. */
636
- declare function extractH2Sections(text: string): string[];
637
- /** Sentence-level edit distance — count distinct add/remove ops between
638
- * two surfaces via a normalised line-by-line set diff. Treats trivial
639
- * whitespace as identical. Exported for tests + consumer-side validators. */
640
- declare function countSentenceEdits(baseline: string, candidate: string): number;
641
-
642
- export { type ParetoResult as A, DEFAULT_RED_TEAM_CORPUS as B, type CampaignRunPlan as C, type Direction as D, type RedTeamCategory as E, type RedTeamFinding as F, type GepaProposerOptions as G, type RedTeamPayload as H, type RedTeamReport as I, crowdingDistance as J, dominates as K, paretoFrontier as L, paretoFrontierWithCrowding as M, redTeamDataset as N, type OpenAutoPrOptions as O, type PlanCampaignRunOptions as P, redTeamReport as Q, type RunOptimizationOptions as R, scalarScore as S, scoreRedTeamOutput as T, toolNamesForRun as U, type RunImprovementLoopResult as a, REFERENCE_EQUIVALENCE_INPUT_LIMITS as b, REFERENCE_EQUIVALENCE_JUDGE_VERSION as c, type ReferenceEquivalenceJudgeInput as d, type ReferenceEquivalenceJudgeOptions as e, type ReferenceEquivalenceJudgeResult as f, type ReferenceEquivalenceScenario as g, type RunCampaignOptions as h, type RunImprovementLoopOptions as i, createReferenceEquivalenceJudge as j, gepaProposer as k, runImprovementLoop as l, runReferenceEquivalenceJudge as m, type RedTeamCase as n, type CampaignRunPlanCell as o, type GepaProposerConstraints as p, type OpenAutoPrResult as q, runCampaign as r, type RunOptimizationResult as s, countSentenceEdits as t, defaultRenderDiff as u, extractH2Sections as v, openAutoPr as w, planCampaignRun as x, runOptimization as y, type Objective as z };