@tangle-network/agent-eval 0.173.2 → 0.174.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (128) hide show
  1. package/CHANGELOG.md +40 -0
  2. package/dist/{proposal-findings-bko3GGy-.js → abort-signal-CtzAM_sJ.js} +11 -11
  3. package/dist/abort-signal-CtzAM_sJ.js.map +1 -0
  4. package/dist/adapters/http.d.ts +2 -2
  5. package/dist/agent-profile-_xPxqVJt.d.ts +488 -0
  6. package/dist/agent-profile-_xPxqVJt.d.ts.map +1 -0
  7. package/dist/analyst/index.d.ts +7 -9
  8. package/dist/analyst/index.d.ts.map +1 -1
  9. package/dist/analyst/index.js +8 -8
  10. package/dist/{benchmark-C4wk_Sjr.js → benchmark-DQKzykkO.js} +2 -2
  11. package/dist/{benchmark-C4wk_Sjr.js.map → benchmark-DQKzykkO.js.map} +1 -1
  12. package/dist/{benchmark-command-CS6gVHVq.js → benchmark-command-mZIlR-ra.js} +13 -13
  13. package/dist/{benchmark-command-CS6gVHVq.js.map → benchmark-command-mZIlR-ra.js.map} +1 -1
  14. package/dist/benchmarks/index.d.ts +3 -4
  15. package/dist/benchmarks/index.d.ts.map +1 -1
  16. package/dist/benchmarks/index.js +3 -3
  17. package/dist/campaign/index.d.ts +5 -9
  18. package/dist/campaign/index.js +7 -7
  19. package/dist/{campaign-B3kPMU8S.js → campaign-BzMSCejE.js} +8 -8
  20. package/dist/{campaign-B3kPMU8S.js.map → campaign-BzMSCejE.js.map} +1 -1
  21. package/dist/cli.js +1 -1
  22. package/dist/{client-DlqdbM7n.d.ts → client-vyYQg3bm.d.ts} +2 -2
  23. package/dist/{client-DlqdbM7n.d.ts.map → client-vyYQg3bm.d.ts.map} +1 -1
  24. package/dist/contract/index.d.ts +9 -10
  25. package/dist/contract/index.js +8 -8
  26. package/dist/{default-registry-B0bKikCb.js → default-registry-CrAp0pYq.js} +4 -4
  27. package/dist/{default-registry-B0bKikCb.js.map → default-registry-CrAp0pYq.js.map} +1 -1
  28. package/dist/{default-registry-BKwc8bN5.d.ts → default-registry-FfNzaUHV.d.ts} +3 -3
  29. package/dist/{default-registry-BKwc8bN5.d.ts.map → default-registry-FfNzaUHV.d.ts.map} +1 -1
  30. package/dist/{define-agent-eval-CY6qdlGV.d.ts → define-agent-eval-V1jQyCDR.d.ts} +102 -11
  31. package/dist/define-agent-eval-V1jQyCDR.d.ts.map +1 -0
  32. package/dist/{define-agent-eval-8h3lXXee.js → define-agent-eval-ox5McL6e.js} +331 -144
  33. package/dist/define-agent-eval-ox5McL6e.js.map +1 -0
  34. package/dist/{dspy-rlm-engine-CF0t2ITD.js → dspy-rlm-engine-Caz2pl4L.js} +3 -3
  35. package/dist/{dspy-rlm-engine-CF0t2ITD.js.map → dspy-rlm-engine-Caz2pl4L.js.map} +1 -1
  36. package/dist/{engine-DhFir3Ys.d.ts → engine-CvW_I72-.d.ts} +2 -2
  37. package/dist/{engine-DhFir3Ys.d.ts.map → engine-CvW_I72-.d.ts.map} +1 -1
  38. package/dist/experiment/index.d.ts +1 -4
  39. package/dist/experiment/index.d.ts.map +1 -1
  40. package/dist/{external-optimizer-process-BwITA9Jp.js → external-optimizer-process-CxnFL1hd.js} +2 -2
  41. package/dist/{external-optimizer-process-BwITA9Jp.js.map → external-optimizer-process-CxnFL1hd.js.map} +1 -1
  42. package/dist/{external-optimizer-subprocess-wBWeoG6A.js → external-optimizer-subprocess-CQi27uEI.js} +2 -2
  43. package/dist/{external-optimizer-subprocess-wBWeoG6A.js.map → external-optimizer-subprocess-CQi27uEI.js.map} +1 -1
  44. package/dist/fuzz.js +1 -1
  45. package/dist/fuzz.js.map +1 -1
  46. package/dist/hosted/index.d.ts +1 -1
  47. package/dist/{index-BQqOjerE.d.ts → index-BTrx5s8m.d.ts} +8 -9
  48. package/dist/index-BTrx5s8m.d.ts.map +1 -0
  49. package/dist/{index-D0Db5X-4.d.ts → index-Bn-nlnSV.d.ts} +4 -4
  50. package/dist/{index-D0Db5X-4.d.ts.map → index-Bn-nlnSV.d.ts.map} +1 -1
  51. package/dist/index-DKXuBPXf.d.ts +3840 -0
  52. package/dist/index-DKXuBPXf.d.ts.map +1 -0
  53. package/dist/index.d.ts +11 -13
  54. package/dist/index.d.ts.map +1 -1
  55. package/dist/index.js +10 -10
  56. package/dist/{kind-factory-gP6lDySe.js → kind-factory-BLvL-E44.js} +2 -2
  57. package/dist/{kind-factory-gP6lDySe.js.map → kind-factory-BLvL-E44.js.map} +1 -1
  58. package/dist/{llm-judge-BfqMFo4h.js → llm-judge-DmNaBrXB.js} +2541 -2435
  59. package/dist/llm-judge-DmNaBrXB.js.map +1 -0
  60. package/dist/{matrix-DGu8KhSs.d.ts → matrix-CJtXz1ky.d.ts} +2 -2
  61. package/dist/{matrix-DGu8KhSs.d.ts.map → matrix-CJtXz1ky.d.ts.map} +1 -1
  62. package/dist/multishot/golden/index.d.ts +1 -1
  63. package/dist/multishot/index.d.ts +2 -2
  64. package/dist/openapi.json +1 -1
  65. package/dist/{produced-state-D91uDvQw.js → produced-state-B8mw6zj9.js} +2 -2
  66. package/dist/{produced-state-D91uDvQw.js.map → produced-state-B8mw6zj9.js.map} +1 -1
  67. package/dist/rl.d.ts +1 -1
  68. package/dist/rl.d.ts.map +1 -1
  69. package/dist/rl.js.map +1 -1
  70. package/dist/{semantic-concept-judge-Dok7_35a.js → semantic-concept-judge-E3s_fEjB.js} +3 -3
  71. package/dist/{semantic-concept-judge-Dok7_35a.js.map → semantic-concept-judge-E3s_fEjB.js.map} +1 -1
  72. package/dist/{skillopt-optimization-method-DDw3v3gA.js → skillopt-optimization-method-f7399oGb.js} +5 -5
  73. package/dist/{skillopt-optimization-method-DDw3v3gA.js.map → skillopt-optimization-method-f7399oGb.js.map} +1 -1
  74. package/dist/statistical-heldout-Cqb73yE9.d.ts +1127 -0
  75. package/dist/statistical-heldout-Cqb73yE9.d.ts.map +1 -0
  76. package/dist/{store-otlp-Dow0pk_5.js → store-otlp-DV_H2HDu.js} +2 -2
  77. package/dist/{store-otlp-Dow0pk_5.js.map → store-otlp-DV_H2HDu.js.map} +1 -1
  78. package/dist/{store-tool-spans-CCZNsihA.d.ts → store-tool-spans-4o55ABER.d.ts} +3 -3
  79. package/dist/{store-tool-spans-CCZNsihA.d.ts.map → store-tool-spans-4o55ABER.d.ts.map} +1 -1
  80. package/dist/{store-tool-spans-CeNj_m2L.js → store-tool-spans-B9tjys_h.js} +3 -3
  81. package/dist/{store-tool-spans-CeNj_m2L.js.map → store-tool-spans-B9tjys_h.js.map} +1 -1
  82. package/dist/supervisor-run/index.d.ts.map +1 -1
  83. package/dist/supervisor-run/index.js +25 -7
  84. package/dist/supervisor-run/index.js.map +1 -1
  85. package/dist/{task-failure-attributes-CZjZeBsY.js → task-failure-attributes-CUy9mkIY.js} +2 -2
  86. package/dist/{task-failure-attributes-CZjZeBsY.js.map → task-failure-attributes-CUy9mkIY.js.map} +1 -1
  87. package/dist/{tool-groups-Cp4Xdzrp.d.ts → tool-groups-DAe1t6zb.d.ts} +2 -2
  88. package/dist/tool-groups-DAe1t6zb.d.ts.map +1 -0
  89. package/dist/trace-repair/index.d.ts +1 -1
  90. package/dist/traces.d.ts +2 -2
  91. package/dist/traces.js +4 -4
  92. package/dist/{types-Ba5UQyVD.d.ts → types-BJz2CPTM.d.ts} +2 -2
  93. package/dist/{types-Ba5UQyVD.d.ts.map → types-BJz2CPTM.d.ts.map} +1 -1
  94. package/dist/{types-CiWITkGo.js → types-DQ0e2E7y.js} +2 -2
  95. package/dist/types-DQ0e2E7y.js.map +1 -0
  96. package/dist/{types-BDV4PiMR.d.ts → types-Dd1ejaeI.d.ts} +2 -2
  97. package/dist/{types-BDV4PiMR.d.ts.map → types-Dd1ejaeI.d.ts.map} +1 -1
  98. package/docs/campaign-proposers.md +42 -0
  99. package/package.json +1 -1
  100. package/dist/agent-profile-B9_GGsG8.d.ts +0 -84
  101. package/dist/agent-profile-B9_GGsG8.d.ts.map +0 -1
  102. package/dist/backend-integrity-CeuTgqsd.d.ts +0 -280
  103. package/dist/backend-integrity-CeuTgqsd.d.ts.map +0 -1
  104. package/dist/benchmark-BjLGkfnN.d.ts +0 -236
  105. package/dist/benchmark-BjLGkfnN.d.ts.map +0 -1
  106. package/dist/define-agent-eval-8h3lXXee.js.map +0 -1
  107. package/dist/define-agent-eval-CY6qdlGV.d.ts.map +0 -1
  108. package/dist/external-optimizer-contracts-CQCpyrIL.d.ts +0 -172
  109. package/dist/external-optimizer-contracts-CQCpyrIL.d.ts.map +0 -1
  110. package/dist/heldout-gate-Df5hsqmm.d.ts +0 -453
  111. package/dist/heldout-gate-Df5hsqmm.d.ts.map +0 -1
  112. package/dist/index-BQqOjerE.d.ts.map +0 -1
  113. package/dist/index-CFDffsKz.d.ts +0 -1135
  114. package/dist/index-CFDffsKz.d.ts.map +0 -1
  115. package/dist/llm-judge-BfqMFo4h.js.map +0 -1
  116. package/dist/power-preflight-Ptse_Kq7.d.ts +0 -117
  117. package/dist/power-preflight-Ptse_Kq7.d.ts.map +0 -1
  118. package/dist/pre-registration-BoI4ucR3.d.ts +0 -592
  119. package/dist/pre-registration-BoI4ucR3.d.ts.map +0 -1
  120. package/dist/promotion-policy-CvMda3kU.d.ts +0 -134
  121. package/dist/promotion-policy-CvMda3kU.d.ts.map +0 -1
  122. package/dist/proposal-findings-bko3GGy-.js.map +0 -1
  123. package/dist/provenance-CRY67X50.d.ts +0 -1995
  124. package/dist/provenance-CRY67X50.d.ts.map +0 -1
  125. package/dist/statistical-heldout-DTyB_6-1.d.ts +0 -295
  126. package/dist/statistical-heldout-DTyB_6-1.d.ts.map +0 -1
  127. package/dist/tool-groups-Cp4Xdzrp.d.ts.map +0 -1
  128. package/dist/types-CiWITkGo.js.map +0 -1
@@ -1,1995 +0,0 @@
1
- import { c as CostLedgerHandle, f as CostLedgerSummary, m as CostReceipt, o as CostLedger, p as CostProvenance } from "./cost-ledger-DbQdN3nO.js";
2
- import { a as RunRecord } from "./run-record-DTv1MdjK.js";
3
- import { _ as ProposalFinding } from "./types-DN2WdT5S.js";
4
- import { p as ChatClient } from "./types-gvRsyJLh.js";
5
- import { B as ScoredSurfaceOutcome, C as JudgeDimension, H as SurfaceProposer, N as ParetoParent, R as Scenario, S as JudgeConfig, _ as GateDecision, a as CampaignResult, b as GenerationRecord, c as CampaignTraceWriter, d as DispatchContext, f as DispatchFn, g as GateContribution, j as MutableSurface, k as LabeledScenarioStore, p as Gate, r as CampaignCellResult, v as GateResult, y as GenerationCandidate } from "./types-Ba5UQyVD.js";
6
- import { m as LedgerTrustedHeadRemoval, p as LedgerTrustedHead } from "./index-D-UdhAmg.js";
7
- import { r as LedgerHash } from "./canonical-CFpojCN5.js";
8
- import { _ as ExternalTextCandidate, g as ExternalOptimizerWireCounts, o as ExternalOptimizerEvaluationObservation } from "./external-optimizer-contracts-CQCpyrIL.js";
9
- import { r as DatasetScenario, t as Dataset } from "./dataset-DQqhOCPt.js";
10
- import { g as TraceSpanEvent, t as HostedClient } from "./client-DlqdbM7n.js";
11
- import { z } from "zod";
12
- //#region src/campaign/storage.d.ts
13
- /**
14
- * `CampaignStorage` — the filesystem seam `runCampaign` writes through
15
- * (run/cell dirs, the resumability cache, per-cell artifacts, trace spans).
16
- *
17
- * The default (`fsCampaignStorage`) is the Node filesystem — identical
18
- * behavior to the inline `node:fs` calls it replaces, so existing CLI
19
- * consumers are unaffected. `inMemoryCampaignStorage` keeps everything in a
20
- * `Map`, so the substrate runs in environments WITHOUT a filesystem
21
- * (Cloudflare Workers, Deno Deploy, other edge runtimes) — the campaign
22
- * still produces its `CampaignResult` (cells + aggregates) in memory;
23
- * artifacts/traces simply aren't persisted to disk.
24
- *
25
- * Paths are opaque keys to the in-memory adapter — it does not parse them,
26
- * so the same `join(...)`-built paths work unchanged across both adapters.
27
- */
28
- interface CampaignStorage {
29
- /** Ensure a directory exists (recursive). No-op for in-memory. */
30
- ensureDir(dir: string): void;
31
- /** Does this path exist (as a written file or an ensured dir)? */
32
- exists(path: string): boolean;
33
- /** Read a UTF-8 file; `undefined` when missing or unreadable. */
34
- read(path: string): string | undefined;
35
- /** Write a file (string or bytes). Parent dir is assumed ensured. */
36
- write(path: string, content: string | Uint8Array): void;
37
- /** Append only when the current UTF-8 byte length matches `expectedBytes`.
38
- * Returns the new length, or undefined when another writer won. */
39
- append(path: string, content: string, expectedBytes: number): number | undefined;
40
- }
41
- /** Node-filesystem storage — the default. Lazily requires `node:fs` so the
42
- * module imports cleanly in non-Node runtimes (where the caller passes
43
- * `inMemoryCampaignStorage` instead and never constructs this).
44
- *
45
- * `createRequire(import.meta.url)` is the ESM-native lazy require — a bare
46
- * `require` is a ReferenceError under `"type": "module"`, which is exactly
47
- * the shape this package publishes. */
48
- declare function fsCampaignStorage(): CampaignStorage;
49
- /** In-memory storage for filesystem-less runtimes. Artifacts + trace spans
50
- * live in a `Map` for the duration of the run; the `CampaignResult` is
51
- * fully populated, but nothing is persisted to disk. */
52
- declare function inMemoryCampaignStorage(): CampaignStorage;
53
- /** Open the durable spend account stored beside a logical run. */
54
- declare function createRunCostLedger(input: {
55
- storage: CampaignStorage;
56
- runDir: string;
57
- costCeilingUsd?: number;
58
- /** Set false for read-only inspection that must not create the run directory. */
59
- ensureRunDir?: boolean;
60
- }): CostLedger;
61
- //#endregion
62
- //#region src/campaign/cell-schedule.d.ts
63
- declare function buildCellSchedule<TScenario extends Scenario>(scenarios: TScenario[], seed: number, reps: number): Array<{
64
- scenario: TScenario;
65
- rep: number;
66
- cellId: string;
67
- cellSeed: number;
68
- }>;
69
- type CellScheduleSlot<TScenario extends Scenario> = ReturnType<typeof buildCellSchedule<TScenario>>[number];
70
- declare function cellCachePath(runDir: string, cellId: string): string;
71
- //#endregion
72
- //#region src/campaign/cell-cache.d.ts
73
- type CacheIssueReason = 'missing' | 'manifest-mismatch' | 'cell-mismatch' | 'missing-cost-provenance' | 'invalid-cost-provenance' | 'invalid-cost-receipts' | 'corrupt';
74
- type CacheRead<TArtifact> = {
75
- status: 'hit';
76
- cell: CampaignCellResult<TArtifact>;
77
- } | {
78
- status: 'miss';
79
- reason: CacheIssueReason;
80
- };
81
- declare function readCachedCell<TArtifact>(args: {
82
- storage: CampaignStorage;
83
- cachePath: string;
84
- cellId: string;
85
- manifestHash: string;
86
- }): CacheRead<TArtifact>;
87
- //#endregion
88
- //#region src/campaign/plan-campaign-run.d.ts
89
- interface CampaignRunPlanCell {
90
- cellId: string;
91
- scenarioId: string;
92
- rep: number;
93
- seed: number;
94
- cachePath: string;
95
- status: 'cached' | 'run' | 'blocked';
96
- reason?: CacheIssueReason | 'resumable-off';
97
- }
98
- interface CampaignRunPlan {
99
- manifestHash: string;
100
- splitDigest: `sha256:${string}`;
101
- totalCells: number;
102
- cellsCached: number;
103
- cellsBlocked: number;
104
- cellsToRun: number;
105
- cells: CampaignRunPlanCell[];
106
- }
107
- interface PlanCampaignRunOptions<TScenario extends Scenario, TArtifact> {
108
- scenarios: TScenario[];
109
- dispatch?: DispatchFn<TScenario, TArtifact>;
110
- dispatchRef?: string;
111
- judges?: JudgeConfig<TArtifact, TScenario>[];
112
- seed?: number;
113
- reps?: number;
114
- resumable?: boolean;
115
- /** See RunCampaignOptions.rerunInvalidCachedCells. */
116
- rerunInvalidCachedCells?: boolean;
117
- runDir: string;
118
- /** Subject repo for the shared run-dir root (see RunCampaignOptions.repo). */
119
- repo?: string;
120
- storage?: CampaignStorage;
121
- /** Spend account used to validate cached receipt identities. */
122
- costLedger?: CostLedgerHandle;
123
- /** Receipt tags used by the campaign that produced the cached cells. */
124
- costTags?: Readonly<Record<string, string>>;
125
- }
126
- /**
127
- * Plan a campaign WITHOUT dispatching: computes the manifest hash and the per-cell
128
- * run-vs-cached schedule so callers can preview cost and resumability before spending.
129
- */
130
- declare function planCampaignRun<TScenario extends Scenario, TArtifact>(opts: PlanCampaignRunOptions<TScenario, TArtifact>): CampaignRunPlan;
131
- //#endregion
132
- //#region src/campaign/run-campaign.d.ts
133
- interface RunCampaignOptions<TScenario extends Scenario, TArtifact> {
134
- scenarios: TScenario[];
135
- dispatch: DispatchFn<TScenario, TArtifact>;
136
- /** Abort active dispatches when the owning operation is cancelled. */
137
- signal?: AbortSignal;
138
- /**
139
- * Stable identity for the dispatch behavior, included in the manifest/cache
140
- * key. Set this when the same function name can run different models,
141
- * prompts, tools, or external config.
142
- */
143
- dispatchRef?: string;
144
- judges?: JudgeConfig<TArtifact, TScenario>[];
145
- /** Required for reproducibility. Default 42. */
146
- seed?: number;
147
- /** Per-scenario replicates for CI bands. Default 1; raise to 5+ for
148
- * bootstrap-tight intervals on critical eval. */
149
- reps?: number;
150
- /** When true (default), completed cells are cached by
151
- * (manifestHash, scenarioId, rep, generation). Re-runs skip cached cells. */
152
- resumable?: boolean;
153
- /**
154
- * Optional explicit cell selection. The campaign manifest and split digest
155
- * still describe the complete declared scenario × replicate design; this
156
- * only limits the rows executed by this invocation.
157
- */
158
- cellFilter?: (input: {
159
- scenario: TScenario;
160
- rep: number;
161
- }) => boolean;
162
- /**
163
- * Reuse a cached cell that has an error instead of dispatching it again.
164
- * The default retries failed cells, preserving normal campaign behaviour.
165
- */
166
- reuseFailedCells?: boolean;
167
- /**
168
- * Explicitly rerun only cached cells whose saved result is unreadable or
169
- * has missing/invalid cost provenance. Valid cached cells remain reusable.
170
- * Default false refuses to begin work when any such cache entry exists.
171
- */
172
- rerunInvalidCachedCells?: boolean;
173
- /** Optional store — when present, every artifact + judge score is captured
174
- * with the configured `captureSource`. Capture is default ON; pass `'off'`
175
- * to disable. */
176
- labeledStore?: LabeledScenarioStore | 'off';
177
- captureSource?: 'production-trace' | 'eval-run' | 'manual' | 'red-team' | 'synthetic';
178
- captureSourceVersionHash?: string;
179
- /** Hard spend cap. Each paid call reserves its enforced maximum before dispatch. */
180
- costCeiling?: number;
181
- /** Shared spend account. Improvement loops pass one ledger through every
182
- * campaign so the ceiling and returned total are run-wide. */
183
- costLedger?: CostLedgerHandle;
184
- /** Attribution label for receipts recorded by this campaign. */
185
- costPhase?: string;
186
- /** Additional immutable receipt tags supplied by an owning workflow. */
187
- costTags?: Readonly<Record<string, string>>;
188
- /** Max concurrent cells. Default 2. */
189
- maxConcurrency?: number;
190
- /**
191
- * Stop after the first dispatch or judge error. The failed cell is persisted
192
- * before active sibling cells are aborted and drained, then the campaign
193
- * rejects with the exact error thrown by that dispatch or judge.
194
- * Default false preserves the normal behavior of returning failed cells and
195
- * continuing the remaining schedule.
196
- * With `cellRetry`, a retryable failure is not an error yet: this abort
197
- * fires only when a cell's final attempt fails.
198
- */
199
- abortOnCellError?: boolean;
200
- /**
201
- * Opt-in bounded in-run retry of a failed cell. Absent by default: a failed
202
- * cell is final on its first attempt. A retried attempt re-runs the SAME
203
- * slot — same `cellId`, same `seed`, same cost tags — so the schedule,
204
- * manifest, and pairing are unchanged. An attempt that failed because the
205
- * campaign was cancelled is never retried.
206
- */
207
- cellRetry?: CampaignCellRetryPolicy;
208
- /**
209
- * Per-cell dispatch deadline in ms. A `dispatch` that neither resolves nor
210
- * rejects within this window is a hang (a stalled model request, an
211
- * exhausted runtime resource, a backend that never closes its stream). When
212
- * set, the cell's `ctx.signal` is aborted. A dispatch that stops is recorded
213
- * as an error (`dispatch exceeded <N>ms`). A dispatch that ignores
214
- * cancellation rejects the campaign without publishing incomplete cost data.
215
- * `undefined`/`0` means unbounded.
216
- */
217
- dispatchTimeoutMs?: number;
218
- /**
219
- * Time allowed for an aborted dispatch and its paid calls to stop before the
220
- * campaign rejects without producing a result. Default 5 seconds.
221
- */
222
- dispatchShutdownTimeoutMs?: number;
223
- /** Required: where artifacts + traces land. A bare name (not an absolute path)
224
- * resolves to the shared `~/.tangle/traces/<repo>/runs/<name>` root so run
225
- * bundles never pollute a repo working tree. Pass an absolute path to override. */
226
- runDir: string;
227
- /** Subject repo for the shared run-dir root (defaults to the CWD basename).
228
- * Only consulted when `runDir` is a bare name. */
229
- repo?: string;
230
- /** Tracing posture. Default is the substrate's `FileSystemTraceStore` rooted
231
- * at `<runDir>/traces/`. `'off'` disables capture entirely — substrate
232
- * refuses this when the caller wires `autoOnPromote !== 'none'`. */
233
- tracing?: 'on' | 'off';
234
- /**
235
- * Per-cell usage expectation — the early, fine-grained sibling of the
236
- * batch `assertRealBackend` guard. A cell that produced an artifact (no
237
- * error) but reported `costUsd === 0` AND zero tokens is a stub: the
238
- * dispatch never reported LLM activity via `ctx.cost`. Modes:
239
- * - `'warn'` (default) — log the offending cell loudly, keep going.
240
- * - `'assert'` — throw `BackendIntegrityError` on the first such cell
241
- * (fail-fast; recommended for CI campaigns expecting real LLM calls).
242
- * - `'off'` — no check (replay / deterministic-only / offline analysis).
243
- */
244
- expectUsage?: 'assert' | 'warn' | 'off';
245
- /** Test seam — override the wall clock for deterministic tests. */
246
- now?: () => Date;
247
- /** Test seam — override per-cell trace writer factory. */
248
- buildTraceWriter?: (cellId: string, dir: string) => CampaignTraceWriter;
249
- /** Storage backend for run/cell dirs, the resumability cache, artifacts,
250
- * and trace spans. Default: the Node filesystem (`fsCampaignStorage`).
251
- * Pass `inMemoryCampaignStorage()` to run in a filesystem-less runtime
252
- * (Cloudflare Workers, Deno, edge) — the `CampaignResult` is still
253
- * produced; artifacts/traces just aren't persisted to disk. */
254
- storage?: CampaignStorage;
255
- /**
256
- * Optional per-cell placement strategy. Returns an opaque string the
257
- * substrate forwards as `ctx.placement` to the Dispatch — placement-aware
258
- * Dispatches (e.g. `httpDispatch` from `/adapters/http`) use it to route
259
- * each cell to the right worker, region, or sandbox. When unset, every
260
- * cell receives `ctx.placement = undefined` and behaves identically to
261
- * the in-process case.
262
- *
263
- * @example
264
- * cellPlacement: ({ scenario }) => scenario.tags?.includes('eu') ? 'eu-west' : 'us-east'
265
- */
266
- cellPlacement?: (input: {
267
- scenario: TScenario;
268
- rep: number;
269
- generation?: number;
270
- }) => string | undefined;
271
- }
272
- /** Durable `<cell>/failure-receipt.json` written before a failed cell can
273
- * trigger campaign-wide cancellation. The cell records dispatch measurements;
274
- * `cost` covers every settled agent and judge call attributed to this exact run
275
- * attempt. */
276
- interface CampaignCellFailureReceipt<TArtifact = unknown> {
277
- schemaVersion: 1;
278
- runAttemptId: string;
279
- recordedAt: string;
280
- failure: {
281
- stage: 'dispatch' | 'judge';
282
- judge?: string;
283
- error: {
284
- name: string;
285
- message: string;
286
- stack?: string;
287
- };
288
- };
289
- cell: CampaignCellResult<TArtifact>;
290
- cost: CostLedgerSummary;
291
- }
292
- /**
293
- * Bounded in-run retry of failed cells. Every attempt dispatches the same
294
- * slot and charges the shared cost ledger, so the final cell's `costUsd`,
295
- * `tokenUsage`, and `costCallIds` cover all attempts. Each retried attempt
296
- * keeps its failure receipt at `<cell>/failure-receipt.attempt-<n>.json`; a
297
- * final failed attempt keeps the usual `<cell>/failure-receipt.json`. The
298
- * final cell records the retry count as `retryAttempts`. Artifacts and trace
299
- * spans written by a later attempt replace those of the retried attempt; the
300
- * per-attempt failure receipts are the durable evidence.
301
- */
302
- interface CampaignCellRetryPolicy {
303
- /** Total attempts per cell, including the first. A positive safe integer. */
304
- attempts: number;
305
- /** Decides whether a failed attempt is dispatched again. Receives the
306
- * receipt's `failure` record. `transientDispatchFailure()` is the
307
- * ready-made predicate for infrastructure hiccups (502/503/504, dropped
308
- * streams, admission rejections). */
309
- retryable: (failure: CampaignCellFailureReceipt['failure']) => boolean;
310
- }
311
- /**
312
- * Core campaign orchestrator: fan scenarios through dispatch, score with judges, aggregate bootstrap CIs, and persist reproducible `CampaignResult` records.
313
- */
314
- declare function runCampaign<TScenario extends Scenario, TArtifact>(opts: RunCampaignOptions<TScenario, TArtifact>): Promise<CampaignResult<TArtifact, TScenario>>;
315
- //#endregion
316
- //#region src/campaign/presets/run-eval.d.ts
317
- interface RunEvalOptions<TScenario extends Scenario, TArtifact> extends Omit<RunCampaignOptions<TScenario, TArtifact>, 'runDir'> {
318
- runDir: string;
319
- }
320
- /**
321
- * Simplest evaluation preset: run scenarios through dispatch, score with judges, and return a `CampaignResult` — no optimizer, no gate, no PR.
322
- */
323
- declare function runEval<TScenario extends Scenario, TArtifact>(opts: RunEvalOptions<TScenario, TArtifact>): Promise<CampaignResult<TArtifact, TScenario>>;
324
- //#endregion
325
- //#region src/campaign/auto-pr.d.ts
326
- interface OpenAutoPrOptions<TArtifact, TScenario extends Scenario> {
327
- /** Campaign result to attach to the PR. */
328
- result: CampaignResult<TArtifact, TScenario>;
329
- /** Gate verdict explaining the promotion. Substrate refuses to open a PR
330
- * when `gate.decision !== 'ship'` — fails loud. */
331
- gate: GateResult;
332
- /** Promoted surface diff — typically the new system prompt addendum or
333
- * full profile diff. Substrate writes it as the PR body. */
334
- promotedDiff: string;
335
- /** GH owner/repo target (e.g., `tangle-network/gtm-agent`). */
336
- ghOwner: string;
337
- ghRepo: string;
338
- /** Branch name for the PR. Default `auto/<manifestHash[:12]>`. */
339
- branch?: string;
340
- /** PR title. Default includes manifest hash. */
341
- title?: string;
342
- /** Whether to actually open the PR or just dry-run. Default reads
343
- * `GH_AUTO_PR_TOKEN` env — present = open, absent = dry-run. */
344
- dryRun?: boolean;
345
- /** Test seam — substitute `gh pr create` invocation. */
346
- ghExec?: (args: string[]) => {
347
- stdout: string;
348
- stderr: string;
349
- status: number;
350
- };
351
- }
352
- interface OpenAutoPrResult {
353
- opened: boolean;
354
- prUrl?: string;
355
- dryRun: boolean;
356
- reason: string;
357
- }
358
- /**
359
- * Open a GitHub PR for a gate-approved surface promotion, attaching the manifest hash, gate verdict, and diff as the PR body.
360
- */
361
- declare function openAutoPr<TArtifact, TScenario extends Scenario>(options: OpenAutoPrOptions<TArtifact, TScenario>): OpenAutoPrResult;
362
- //#endregion
363
- //#region src/campaign/parent-selection.d.ts
364
- /** Search state supplied to one parent-selection call. */
365
- interface ParentSelectionContext {
366
- /** Non-dominated scored surfaces across the whole run so far, including the
367
- * baseline (`generation: -1`). Never empty. */
368
- readonly frontier: ReadonlyArray<ParetoParent>;
369
- /** Measured result of the global incumbent, the promotion bar. Under the
370
- * default `selectionRankKey` the incumbent is always on `frontier`. */
371
- readonly incumbent: ScoredSurfaceOutcome;
372
- /** Every completed generation so far. */
373
- readonly history: ReadonlyArray<GenerationRecord>;
374
- /** Index of the generation about to propose. */
375
- readonly generation: number;
376
- }
377
- /** Chooses the surface the next generation mutates. Returns one frontier
378
- * parent; `runOptimization` refuses a parent it has not measured to
379
- * completion or whose surface does not match its `surfaceHash`. */
380
- type ParentSelector = (ctx: ParentSelectionContext) => ParetoParent;
381
- interface CrowdedFrontierParentOptions {
382
- /** Integer seed for the per-generation draw. The same seed, frontier, and
383
- * generation index select the same parent. */
384
- seed: number;
385
- }
386
- /**
387
- * NSGA-II crowded tournament selection over the frontier. Each generation
388
- * draws two distinct frontier members with a PRNG seeded from `seed` and the
389
- * generation index, and keeps the one with the larger crowding distance (more
390
- * isolated on the frontier). Boundary parents carry infinite distance, so a
391
- * boundary parent always beats an interior one. A tie on distance falls back
392
- * to the higher mean composite, then to the smaller surface hash. A frontier
393
- * of one member returns that member.
394
- */
395
- declare function crowdedFrontierParent(options: CrowdedFrontierParentOptions): ParentSelector;
396
- //#endregion
397
- //#region src/campaign/search-ledger.d.ts
398
- declare const SEARCH_LEDGER_SCHEMA: 'tangle.search-ledger.v1';
399
- type SearchLedgerHash = LedgerHash;
400
- type SearchSurfaceKind = 'prompt' | 'tool-contract' | 'runtime-config' | 'memory' | 'knowledge' | 'agent-profile' | 'code' | 'deployment';
401
- /** Content-addressed artifact or receipt. Mutable paths are locators only; the
402
- * digest and byte length bind the exact bytes used by the search. */
403
- interface SearchArtifactRef {
404
- role: string;
405
- uri: string;
406
- sha256: SearchLedgerHash;
407
- byteLength: number;
408
- }
409
- /** Repository, dataset, or package source pinned to an immutable commit or
410
- * content digest. Branches, tags, and bare package versions are rejected. */
411
- interface SearchSourceRef {
412
- uri: string;
413
- revision: string;
414
- }
415
- interface SearchModelIdentity {
416
- provider: string;
417
- snapshot: string;
418
- }
419
- interface SearchCandidateSurface {
420
- surfaceId: string;
421
- kind: SearchSurfaceKind;
422
- artifact: SearchArtifactRef;
423
- }
424
- interface SearchCandidateLineage {
425
- /** Existing `LineageNode.id`; this ledger references rather than embeds it. */
426
- lineageNodeId: string;
427
- parentCandidateIds: string[];
428
- generation: number;
429
- proposer: string;
430
- proposerSource: SearchSourceRef;
431
- }
432
- type SearchOperationKind = 'candidate-generation' | 'analysis' | 'selection' | 'judge' | 'other';
433
- interface SearchPlannedTask {
434
- taskId: string;
435
- source: SearchSourceRef;
436
- benchmark: SearchSourceRef;
437
- /** Maximum transport attempts for this task and candidate. Only an explicit
438
- * passed/failed outcome satisfies the planned denominator. */
439
- maxAttempts: number;
440
- }
441
- interface SearchPlannedOperation {
442
- operationId: string;
443
- kind: SearchOperationKind;
444
- }
445
- interface SearchCandidateSlot {
446
- slotId: string;
447
- /** Planned candidate-generation call that must either produce this slot or
448
- * fail before the slot can be closed. Several slots may share one batched call. */
449
- generationOperationId: string;
450
- }
451
- interface SearchPlan {
452
- /** Stable slots and their proposer calls are frozen before search begins. */
453
- candidateSlots: SearchCandidateSlot[];
454
- /** Every task applies to every successfully registered candidate. */
455
- tasks: SearchPlannedTask[];
456
- /** Non-task spend slots: proposal, analysis, selection, extra judges, etc. */
457
- operations: SearchPlannedOperation[];
458
- }
459
- type SearchTokenAccounting = {
460
- status: 'known';
461
- inputTokens: number;
462
- outputTokens: number;
463
- cachedTokens: number;
464
- } | {
465
- status: 'unknown';
466
- reason: string;
467
- };
468
- type SearchCostAccounting = {
469
- status: 'known';
470
- usd: number;
471
- source: 'provider' | 'pricing-table' | 'free';
472
- } | {
473
- status: 'unknown';
474
- /** Known spend may still be a lower bound when one call was unpriced. */
475
- knownLowerBoundUsd: number;
476
- reason: string;
477
- };
478
- interface SearchAttemptAccounting {
479
- tokens: SearchTokenAccounting;
480
- cost: SearchCostAccounting;
481
- }
482
- interface SearchFailureReason {
483
- code: string;
484
- message: string;
485
- }
486
- type SearchTaskOutcome = {
487
- status: 'passed';
488
- score: number;
489
- metrics: Record<string, number>;
490
- } | {
491
- status: 'failed';
492
- score: number;
493
- metrics: Record<string, number>;
494
- failure: SearchFailureReason;
495
- } | {
496
- status: 'errored';
497
- metrics: Record<string, number>;
498
- error: SearchFailureReason & {
499
- retryable: boolean;
500
- };
501
- };
502
- type SearchSurfaceEffect = {
503
- status: 'measured';
504
- metric: string;
505
- baselineValue: number;
506
- candidateValue: number;
507
- delta: number;
508
- } | {
509
- status: 'not-measured';
510
- reason: string;
511
- };
512
- /** Per-attempt proof that a declared candidate surface was or was not active,
513
- * plus measured effect when the experiment supports attribution. */
514
- interface SearchSurfaceEvidence {
515
- surfaceId: string;
516
- fired: boolean;
517
- firingCount: number;
518
- effect: SearchSurfaceEffect;
519
- evidence: SearchArtifactRef[];
520
- }
521
- interface SearchLedgerEventBase {
522
- eventId: string;
523
- occurredAt: string;
524
- artifacts: SearchArtifactRef[];
525
- }
526
- interface SearchPlannedEvent extends SearchLedgerEventBase {
527
- kind: 'search-planned';
528
- plan: SearchPlan;
529
- }
530
- /** Additional candidate slots and operations for a search whose length is not
531
- * known when it starts. The plan stays the first event and the planned task
532
- * denominator stays frozen: extending tasks would retroactively reopen
533
- * candidates that already closed theirs. */
534
- interface SearchPlanExtendedEvent extends SearchLedgerEventBase {
535
- kind: 'search-plan-extended';
536
- extension: {
537
- candidateSlots: SearchCandidateSlot[];
538
- operations: SearchPlannedOperation[];
539
- };
540
- }
541
- interface SearchCandidateRegisteredEvent extends SearchLedgerEventBase {
542
- kind: 'candidate-registered';
543
- slotId: string;
544
- generationOperationId: string;
545
- candidateId: string;
546
- lineage: SearchCandidateLineage;
547
- surfaces: SearchCandidateSurface[];
548
- }
549
- interface SearchCandidateSlotClosedEvent extends SearchLedgerEventBase {
550
- kind: 'candidate-slot-closed';
551
- slotId: string;
552
- generationOperationId: string;
553
- reason: SearchFailureReason;
554
- }
555
- interface SearchTaskAttemptedEvent extends SearchLedgerEventBase {
556
- kind: 'task-attempted';
557
- candidateId: string;
558
- runId: string;
559
- attemptIndex: number;
560
- task: {
561
- taskId: string;
562
- source: SearchSourceRef;
563
- };
564
- identity: {
565
- model: SearchModelIdentity;
566
- agent: SearchSourceRef;
567
- benchmark: SearchSourceRef;
568
- };
569
- outcome: SearchTaskOutcome;
570
- accounting: SearchAttemptAccounting;
571
- surfaceEvidence: SearchSurfaceEvidence[];
572
- }
573
- interface SearchOperationRecordedEvent extends SearchLedgerEventBase {
574
- kind: 'search-operation-recorded';
575
- operationId: string;
576
- operationKind: SearchOperationKind;
577
- execution: {
578
- kind: 'model';
579
- model: SearchModelIdentity;
580
- source: SearchSourceRef;
581
- } | {
582
- kind: 'deterministic';
583
- source: SearchSourceRef;
584
- };
585
- outcome: {
586
- status: 'completed';
587
- } | {
588
- status: 'partial';
589
- failure: SearchFailureReason;
590
- } | {
591
- status: 'failed';
592
- failure: SearchFailureReason;
593
- };
594
- accounting: SearchAttemptAccounting;
595
- }
596
- interface SearchCandidateDecidedEvent extends SearchLedgerEventBase {
597
- kind: 'candidate-decided';
598
- candidateId: string;
599
- decision: {
600
- status: 'selected';
601
- } | {
602
- status: 'rejected';
603
- reason: SearchFailureReason;
604
- };
605
- }
606
- interface SearchCompletedEvent extends SearchLedgerEventBase {
607
- kind: 'search-completed';
608
- result: {
609
- status: 'selected';
610
- candidateId: string;
611
- } | {
612
- status: 'all-rejected';
613
- reason: SearchFailureReason;
614
- };
615
- }
616
- type SearchLedgerEvent = SearchPlannedEvent | SearchPlanExtendedEvent | SearchCandidateRegisteredEvent | SearchCandidateSlotClosedEvent | SearchTaskAttemptedEvent | SearchOperationRecordedEvent | SearchCandidateDecidedEvent | SearchCompletedEvent;
617
- interface SearchLedgerEntry {
618
- schema: typeof SEARCH_LEDGER_SCHEMA;
619
- campaignId: string;
620
- sequence: number;
621
- previousHash: SearchLedgerHash | null;
622
- event: SearchLedgerEvent;
623
- entryHash: SearchLedgerHash;
624
- }
625
- type SearchAccountingAudit = {
626
- status: 'known';
627
- inputTokens: number;
628
- outputTokens: number;
629
- cachedTokens: number;
630
- costUsd: number;
631
- } | {
632
- status: 'partial';
633
- knownInputTokens: number;
634
- knownOutputTokens: number;
635
- knownCachedTokens: number;
636
- knownCostUsd: number;
637
- unknownTokenEventIds: string[];
638
- unknownCostEventIds: string[];
639
- };
640
- interface SearchLedgerAudit {
641
- campaignId: string;
642
- eventCount: number;
643
- candidateCount: number;
644
- closedCandidateSlotCount: number;
645
- attemptCount: number;
646
- operationCount: number;
647
- outcomes: {
648
- passed: number;
649
- failed: number;
650
- errored: number;
651
- };
652
- operationOutcomes: {
653
- completed: number;
654
- partial: number;
655
- failed: number;
656
- };
657
- decisions: {
658
- selected: number;
659
- rejected: number;
660
- pending: number;
661
- };
662
- expected: {
663
- candidateSlots: number;
664
- taskOutcomes: number;
665
- operations: number;
666
- missingCandidateSlots: string[];
667
- missingTaskOutcomes: string[];
668
- missingOperations: string[];
669
- };
670
- status: 'in-progress' | 'selected' | 'all-rejected';
671
- selectedCandidateId: string | null;
672
- accounting: SearchAccountingAudit;
673
- headHash: SearchLedgerHash | null;
674
- }
675
- interface SearchLedgerReplay {
676
- entries: SearchLedgerEntry[];
677
- plan: SearchPlannedEvent | null;
678
- /** Appended plan extensions, in ledger order. The effective plan is the
679
- * first plan event merged with these; `audit.expected` counts the merge. */
680
- planExtensions: SearchPlanExtendedEvent[];
681
- candidates: SearchCandidateRegisteredEvent[];
682
- closedCandidateSlots: SearchCandidateSlotClosedEvent[];
683
- attempts: SearchTaskAttemptedEvent[];
684
- operations: SearchOperationRecordedEvent[];
685
- decisions: SearchCandidateDecidedEvent[];
686
- completion: SearchCompletedEvent | null;
687
- audit: SearchLedgerAudit;
688
- }
689
- interface SearchLedgerAppendResult {
690
- entry: SearchLedgerEntry;
691
- /** False when the exact event was already durably present. */
692
- appended: boolean;
693
- replay: SearchLedgerReplay;
694
- }
695
- /** Validate and return a canonical copy. Arrays whose order is not semantic are
696
- * sorted so retries from different processes produce byte-identical events. */
697
- declare function validateSearchLedgerEvent(input: unknown): SearchLedgerEvent;
698
- /**
699
- * How this ledger uses its trusted head — the `(sequence, entryHash)` pin kept
700
- * in the sibling `<path>.head` file that a hash chain needs to prove entries
701
- * were not deleted from the end. `ledger-core/trusted-head.ts` holds the threat
702
- * model.
703
- *
704
- * - `pin` (default): every append records the new head, and a pin that is
705
- * present is verified on every read.
706
- * - `require`: additionally refuses to read a non-empty ledger whose pin is
707
- * gone, so deleting the sibling file cannot downgrade the guarantee. Only for
708
- * ledgers written under `pin` from their first entry.
709
- * - `off`: chain verification only. Truncation to a valid shorter prefix is
710
- * undetectable.
711
- */
712
- type SearchLedgerTrustedHeadMode = 'pin' | 'require' | 'off';
713
- interface OpenSearchLedgerOptions {
714
- path: string;
715
- campaignId: string;
716
- trustedHead?: SearchLedgerTrustedHeadMode;
717
- }
718
- interface SearchLedger {
719
- readonly path: string;
720
- readonly campaignId: string;
721
- /** Sibling file holding this ledger's trusted head. */
722
- readonly trustedHeadPath: string;
723
- append(event: SearchLedgerEvent): Promise<SearchLedgerAppendResult>;
724
- replay(): Promise<SearchLedgerReplay>;
725
- /** The pinned head, or null when this ledger has never been pinned. */
726
- trustedHead(): Promise<LedgerTrustedHead | null>;
727
- /** Pin the current verified head: how a ledger written under `off`, or one
728
- * whose pin file was removed, acquires a pin without rewriting a byte. */
729
- pinTrustedHead(): Promise<LedgerTrustedHead>;
730
- /** Discard this ledger's pin, reporting what was discarded. Deleting or
731
- * rebuilding the ledger file leaves a pin naming history the file no longer
732
- * carries, and every later read is refused because that is exactly the
733
- * deletion the pin exists to catch; clearing is the supported way to abandon
734
- * that history on purpose. It gives up the deletion guarantee for every entry
735
- * the pin covered. */
736
- clearTrustedHead(): Promise<LedgerTrustedHeadRemoval>;
737
- }
738
- /** Open a durable filesystem search ledger. Construction performs no I/O; the
739
- * first `append` or `replay` validates the complete existing file. */
740
- declare function openSearchLedger(options: OpenSearchLedgerOptions): SearchLedger;
741
- /** Append-only file-backed search ledger with idempotent writes and replay. */
742
- declare class FileSearchLedger implements SearchLedger {
743
- readonly path: string;
744
- readonly campaignId: string;
745
- readonly trustedHeadPath: string;
746
- private readonly trustedHeadMode;
747
- private readonly journal;
748
- constructor(path: string, campaignId: string, trustedHead?: SearchLedgerTrustedHeadMode);
749
- replay(): Promise<SearchLedgerReplay>;
750
- append(input: SearchLedgerEvent): Promise<SearchLedgerAppendResult>;
751
- trustedHead(): Promise<LedgerTrustedHead | null>;
752
- pinTrustedHead(): Promise<LedgerTrustedHead>;
753
- clearTrustedHead(): Promise<LedgerTrustedHeadRemoval>;
754
- }
755
- //#endregion
756
- //#region src/campaign/search-history-receipt.d.ts
757
- declare const SEARCH_HISTORY_RECEIPT_SCHEMA_VERSION: '1.0.0';
758
- declare const SEARCH_HISTORY_RECEIPT_DIGEST_ALGORITHM: 'rfc8785-sha256';
759
- /** Bounded projection of the canonical replay audit. Exact ids stay in SearchLedger. */
760
- interface SearchHistoryAuditSummary {
761
- readonly campaignId: string;
762
- readonly headHash: SearchLedgerHash | null;
763
- readonly status: SearchLedgerAudit['status'];
764
- readonly selectedCandidateId: string | null;
765
- readonly eventCount: number;
766
- readonly candidateCount: number;
767
- readonly closedCandidateSlotCount: number;
768
- readonly attemptCount: number;
769
- readonly operationCount: number;
770
- readonly expectedCandidateSlots: number;
771
- readonly expectedTaskOutcomes: number;
772
- readonly expectedOperations: number;
773
- readonly missingCandidateSlots: number;
774
- readonly missingTaskOutcomes: number;
775
- readonly missingOperations: number;
776
- readonly pendingDecisions: number;
777
- readonly hasPlan: boolean;
778
- readonly hasCompletion: boolean;
779
- }
780
- /**
781
- * A bounded proof envelope over one canonical SearchLedger replay.
782
- *
783
- * The content-addressed ledger remains the sole rich history. This receipt binds
784
- * its producer/run identity, exact audit digest, bounded completeness summary,
785
- * and its own canonical digest. Consumers needing candidate ids, attempts,
786
- * failures, accounting gaps, or decisions read and replay the canonical ledger.
787
- */
788
- interface SearchHistoryReceipt {
789
- readonly schemaVersion: typeof SEARCH_HISTORY_RECEIPT_SCHEMA_VERSION;
790
- readonly kind: 'search-history-receipt';
791
- readonly digestAlgorithm: typeof SEARCH_HISTORY_RECEIPT_DIGEST_ALGORITHM;
792
- readonly receiptDigest: SearchLedgerHash;
793
- /** Stable producer identity, for example an OptimizationMethod name. */
794
- readonly producerId: string;
795
- /** Concrete optimizer/runtime invocation that produced the ledger. */
796
- readonly runId: string;
797
- readonly ledger: SearchArtifactRef;
798
- /** Digest of the complete SearchLedgerAudit produced by canonical replay. */
799
- readonly auditDigest: SearchLedgerHash;
800
- readonly summary: SearchHistoryAuditSummary;
801
- readonly complete: boolean;
802
- readonly incompleteReasons: readonly string[];
803
- }
804
- interface CreateSearchHistoryReceiptInput {
805
- readonly producerId: string;
806
- readonly runId: string;
807
- /** Content-addressed canonical SearchLedger JSONL artifact. */
808
- readonly ledger: SearchArtifactRef;
809
- /** The result returned by SearchLedger.replay(). */
810
- readonly replay: SearchLedgerReplay;
811
- }
812
- type SearchHistoryPolicy = 'allow-missing' | 'require-complete';
813
- interface SearchHistoryCoverageRow {
814
- readonly producerId: string;
815
- readonly status: 'complete' | 'incomplete' | 'missing';
816
- readonly reasons: readonly string[];
817
- readonly receipt?: SearchHistoryReceipt;
818
- }
819
- interface SearchHistoryCoverage {
820
- readonly policy: SearchHistoryPolicy;
821
- readonly allComplete: boolean;
822
- readonly producers: readonly SearchHistoryCoverageRow[];
823
- }
824
- declare class SearchHistoryRequiredError extends Error {
825
- readonly producerId: string;
826
- readonly reasons: readonly string[];
827
- constructor(producerId: string, reasons: readonly string[]);
828
- }
829
- /** Build a bounded receipt from the projection returned by canonical ledger replay. */
830
- declare function createSearchHistoryReceipt(input: CreateSearchHistoryReceiptInput): SearchHistoryReceipt;
831
- /** Verify the bounded receipt. Full ledger bytes are verified by SearchLedger. */
832
- declare function verifySearchHistoryReceipt(receipt: SearchHistoryReceipt): SearchHistoryReceipt;
833
- /** Prove that a receipt still describes the exact canonical replay supplied. */
834
- declare function assertSearchHistoryMatchesReplay(receipt: SearchHistoryReceipt, replay: SearchLedgerReplay): void;
835
- /** Require a receipt owned by this producer and a terminal, denominator-complete history. */
836
- declare function assertCompleteSearchHistory(producerId: string, receipt: SearchHistoryReceipt | undefined): asserts receipt is SearchHistoryReceipt;
837
- /** Classify one producer's history without treating malformed evidence as absence. */
838
- declare function searchHistoryCoverageRow(producerId: string, receipt: SearchHistoryReceipt | undefined): SearchHistoryCoverageRow;
839
- //#endregion
840
- //#region src/campaign/gepa-candidate-population.d.ts
841
- interface GepaCandidatePopulationSummary {
842
- readonly scope: 'gepa-candidate-population';
843
- readonly path: string;
844
- readonly sha256: `sha256:${string}`;
845
- readonly bytes: number;
846
- readonly runId: string;
847
- readonly candidates: number;
848
- readonly bestIndex: number;
849
- readonly maxCandidates: number;
850
- readonly maxCandidateChars: number;
851
- readonly scenarioIds: readonly string[];
852
- readonly surfaceKind: 'text' | 'components';
853
- }
854
- interface GepaCandidateSelectionScore {
855
- readonly scenarioId: string;
856
- readonly score: number;
857
- }
858
- interface GepaCandidatePopulationCandidate {
859
- /** Zero-based index assigned by the exact GEPA result. */
860
- readonly index: number;
861
- readonly candidate: ExternalTextCandidate;
862
- readonly candidateHash: string;
863
- readonly candidateDigest: `sha256:${string}`;
864
- /** Exact GEPA parent indices. The seed candidate has one null parent. */
865
- readonly parentIndices: readonly (number | null)[];
866
- /** Null means GEPA had no selection score for this candidate. */
867
- readonly aggregateScore: number | null;
868
- readonly selectionScores: readonly GepaCandidateSelectionScore[];
869
- readonly discoveryEvaluationCount: number;
870
- }
871
- interface GepaCandidatePopulationArtifact {
872
- readonly summary: GepaCandidatePopulationSummary;
873
- readonly runId: string;
874
- readonly bestIndex: number;
875
- readonly candidates: readonly GepaCandidatePopulationCandidate[];
876
- }
877
- /**
878
- * Read GEPA's exact candidate graph from the artifact addressed by method provenance.
879
- *
880
- * The reader checks the supplied digest, declared byte count, run identity,
881
- * candidate surfaces, parent graph, selection scores, and configured bounds.
882
- * This proves that the bytes match the supplied summary. The caller remains
883
- * responsible for obtaining that summary from trusted method provenance.
884
- */
885
- declare function readGepaCandidatePopulationArtifact(input: {
886
- summary: GepaCandidatePopulationSummary;
887
- storage?: CampaignStorage;
888
- }): GepaCandidatePopulationArtifact;
889
- //#endregion
890
- //#region src/campaign/search-ledger-recording.d.ts
891
- /** How a search operation executed. The shape the ledger event records. */
892
- type SearchExecutionIdentity = SearchOperationRecordedEvent['execution'];
893
- /** Immutable identities the ledger requires and a campaign cannot infer. */
894
- interface SearchRunIdentity {
895
- /** The agent implementation under optimization. */
896
- agent: SearchSourceRef;
897
- /** The candidate generator: a model call or deterministic code. */
898
- proposer: SearchExecutionIdentity;
899
- /** The code that plans the search and selects its winner. */
900
- search: SearchSourceRef;
901
- /** Model the agent runs. Used only for a cell that reported none. */
902
- model: SearchModelIdentity;
903
- }
904
- interface SearchLedgerBinding {
905
- ledger: SearchLedger;
906
- identity: SearchRunIdentity;
907
- }
908
- /** One proposed candidate, before it is measured. */
909
- interface ProposedSearchCandidate {
910
- surface: MutableSurface;
911
- surfaceHash: string;
912
- label?: string;
913
- }
914
- /** One measured candidate, after its campaign scored. */
915
- interface MeasuredSearchCandidate<TArtifact> {
916
- surface: MutableSurface;
917
- surfaceHash: string;
918
- cells: ReadonlyArray<CampaignCellResult<TArtifact>>;
919
- runDir: string;
920
- /** False when the candidate missed a designed cell. */
921
- coverageComplete: boolean;
922
- }
923
- interface SearchRecorderOptions<TScenario extends Scenario> {
924
- binding: SearchLedgerBinding;
925
- storage: CampaignStorage;
926
- runDir: string;
927
- scenarios: ReadonlyArray<TScenario>;
928
- reps: number;
929
- maxGenerations: number;
930
- populationSize: number;
931
- /** Identity of the exact campaign design; the task benchmark pin. */
932
- splitDigest: `sha256:${string}`;
933
- /** Proposer label recorded on every candidate lineage. */
934
- proposerLabel: string;
935
- costLedger: CostLedgerHandle;
936
- }
937
- /**
938
- * Recorder for one `runOptimization` run: `open()`, then `recordGeneration()`
939
- * and `recordResults()` per generation, then `finish()`.
940
- *
941
- * Every event id is derived from the run, and an id already durable is not
942
- * appended again, so a resumed run continues one ledger instead of conflicting
943
- * with its own history.
944
- */
945
- declare class SearchRecorder<TScenario extends Scenario, TArtifact> {
946
- private readonly opts;
947
- private readonly tasks;
948
- private readonly registered;
949
- private readonly order;
950
- private readonly coverage;
951
- private readonly openSlots;
952
- private readonly durableEventIds;
953
- private lastStampMs;
954
- private proposalReceiptCount;
955
- private constructor();
956
- /** Open the recorder and append the plan. An existing ledger for the same
957
- * run is re-read first, so a resumed run keeps one plan and one lineage. */
958
- static open<TScenario extends Scenario, TArtifact>(opts: SearchRecorderOptions<TScenario>): Promise<SearchRecorder<TScenario, TArtifact>>;
959
- /**
960
- * Record one generation's candidate-generation call and the candidates it
961
- * produced. A proposal larger than the planned population extends the plan
962
- * with the extra slots; a proposal that fills fewer closes the rest.
963
- */
964
- recordGeneration(input: {
965
- generation: number;
966
- parentSurfaceHash: string;
967
- candidates: ReadonlyArray<ProposedSearchCandidate>;
968
- }): Promise<void>;
969
- /** Append one task attempt per designed cell of each candidate campaign. */
970
- recordResults(candidates: ReadonlyArray<MeasuredSearchCandidate<TArtifact>>): Promise<void>;
971
- /**
972
- * Close the search: unreached generations, the selection operation, one
973
- * decision per candidate, then the terminal event.
974
- *
975
- * The terminal event is appended only when canonical replay accounts for the
976
- * whole planned denominator. An interrupted or partly unscored search stays
977
- * `in-progress` and its receipt reports the exact gap, instead of claiming a
978
- * closed search.
979
- */
980
- finish(input: {
981
- winnerSurfaceHash: string;
982
- generationsRun: number;
983
- runId: string;
984
- }): Promise<SearchHistoryReceipt>;
985
- /** Bounded receipt over the exact ledger bytes this run produced. */
986
- receipt(runId: string): Promise<SearchHistoryReceipt>;
987
- /** Read an existing ledger for this run so a resume continues it. */
988
- private hydrate;
989
- private plan;
990
- private registeredSlot;
991
- private remember;
992
- private recordGenerationOperation;
993
- private closeSlot;
994
- /** Spend booked to candidate generation since the previous generation. */
995
- private proposalAccounting;
996
- private cellModel;
997
- private proposalArtifact;
998
- /** Write one canonical evidence document and return its content address. */
999
- private writeArtifact;
1000
- private append;
1001
- /** Non-decreasing ISO stamps; the ledger refuses an event that moves back. */
1002
- private stamp;
1003
- }
1004
- /**
1005
- * Record an optimizer's own candidate graph into the same ledger.
1006
- *
1007
- * A complete optimization method searches inside its own process and reports
1008
- * one artifact when it finishes: the candidate population, with each
1009
- * candidate's parents and its score per selection scenario. This turns that
1010
- * artifact into the canonical event stream, so a first-party method returns
1011
- * the same `SearchHistoryReceipt` the in-process loop returns, and
1012
- * `compareOptimizationMethods({ searchHistoryPolicy: 'require-complete' })`
1013
- * accepts it.
1014
- *
1015
- * A candidate the optimizer left unscored on a planned scenario leaves the
1016
- * planned denominator open, so the receipt reports the gap instead of closing
1017
- * the search.
1018
- */
1019
- declare function recordCandidatePopulationSearch<TScenario extends Scenario>(input: {
1020
- ledger: SearchLedger;
1021
- storage: CampaignStorage;
1022
- runDir: string;
1023
- identity: SearchRunIdentity;
1024
- population: GepaCandidatePopulationArtifact;
1025
- /** Scenarios the optimizer selected on. Must cover the population's ids. */
1026
- scenarios: ReadonlyArray<TScenario>;
1027
- /** Spend the optimizer booked to its own candidate generation. */
1028
- generationAccounting: SearchAttemptAccounting;
1029
- producerId: string;
1030
- runId: string;
1031
- }): Promise<SearchHistoryReceipt>;
1032
- //#endregion
1033
- //#region src/campaign/presets/run-optimization.d.ts
1034
- interface PremeasuredOptimizationBaseline<TArtifact, TScenario extends Scenario> {
1035
- /** Hash of the exact surface that produced `campaign`. */
1036
- surfaceHash: string;
1037
- /** Complete prior measurement reused by identity, including artifactsByPath. */
1038
- campaign: CampaignResult<TArtifact, TScenario>;
1039
- }
1040
- interface RunOptimizationBaseOptions<TScenario extends Scenario, TArtifact> extends Omit<RunCampaignOptions<TScenario, TArtifact>, 'dispatch'> {
1041
- /** Initial mutable surface (typically system prompt or addendum). */
1042
- baselineSurface: MutableSurface;
1043
- /**
1044
- * Complete prior measurement of `baselineSurface`. When present,
1045
- * `runOptimization` validates its surface, scenario split, seed, reps, and
1046
- * normal campaign coverage, then skips the baseline campaign entirely — no
1047
- * dispatch or resumability-cache lookup. Candidate campaigns still run
1048
- * normally. Prior spend remains in the imported campaign aggregates and is
1049
- * not added again to this continuation's CostLedger.
1050
- */
1051
- premeasuredBaseline?: PremeasuredOptimizationBaseline<TArtifact, TScenario>;
1052
- /** Dispatcher that takes the CURRENT surface + scenario → artifact. */
1053
- dispatchWithSurface: (surface: MutableSurface, scenario: TScenario, ctx: Parameters<RunCampaignOptions<TScenario, TArtifact>['dispatch']>[1]) => Promise<TArtifact>;
1054
- /** The candidate-generation strategy. */
1055
- proposer: SurfaceProposer<ProposalFinding>;
1056
- populationSize: number;
1057
- maxGenerations: number;
1058
- /** Candidate campaigns run at once. Default 1. Total concurrent cells are
1059
- * bounded by candidateConcurrency * maxConcurrency. */
1060
- candidateConcurrency?: number;
1061
- /** DEPTH knob forwarded to the proposer's `propose()` — max iterations the
1062
- * agentic generator may take per candidate. */
1063
- maxImprovementShots?: number;
1064
- /** Search or observed-production findings forwarded to candidate generation. */
1065
- findings?: ReadonlyArray<ProposalFinding>;
1066
- /** Per-generation findings producer. Runs once on the BASELINE campaign
1067
- * (as `generation: -1`, the baseline convention) before generation 0
1068
- * proposes — so even a single-generation run proposes with trace context —
1069
- * and then after each generation's candidates are scored with that
1070
- * generation's results; whatever it returns REPLACES `ctx.findings` for the
1071
- * NEXT `propose()`, so the diagnosis is refreshed each round instead
1072
- * of being a static one-shot. Generic by design: the substrate does not
1073
- * import an analyst — the consumer plugs its trace-analyst registry / HALO
1074
- * here (reading the per-candidate `runDir` traces). When absent, findings
1075
- * stay the static `opts.findings`. */
1076
- analyzeGeneration?: (input: {
1077
- generation: number;
1078
- runDir: string;
1079
- candidates: Array<{
1080
- surfaceHash: string;
1081
- campaign: CampaignResult<TArtifact, TScenario>;
1082
- composite: number | null;
1083
- }>;
1084
- history: GenerationRecord[];
1085
- /** Shared run spend account and receipt attribution phase. */
1086
- costLedger?: CostLedgerHandle;
1087
- costPhase?: string;
1088
- }) => Promise<ReadonlyArray<ProposalFinding>>;
1089
- /**
1090
- * Optional override for how the WINNER is selected among coverage-complete
1091
- * candidates (and how the incumbent bar is set). Returns a lexicographic rank
1092
- * key — each element higher-is-better; candidates are ranked by descending key
1093
- * (`compareRankKeys`) and the top must STRICTLY beat the incumbent's key to
1094
- * promote. Defaults to `[campaignMeanComposite(campaign)]`, i.e. the historical
1095
- * scalar-mean ranking (single-element key ⇒ identical behavior).
1096
- *
1097
- * A binary-with-replicates consumer (e.g. swe-arena, whose ship-gate counts an
1098
- * instance resolved only when EVERY replicate resolved) passes a fail-closed
1099
- * key built from the SAME reduction its gate uses, so winner-selection and the
1100
- * ship-gate rank on the identical metric and can never invert — the selector
1101
- * cannot promote a flaky per-cell-mean candidate the gate would reject over a
1102
- * fail-closed candidate the gate would accept. Only the winner CHOICE changes;
1103
- * the descriptive `composite` (mean) on every record and the Pareto objective
1104
- * vectors are untouched, so proposer diversity and reporting are unaffected.
1105
- */
1106
- selectionRankKey?: (campaign: CampaignResult<TArtifact, TScenario>) => number[];
1107
- /**
1108
- * Optional policy for which scored surface the next generation MUTATES.
1109
- * Absent, every generation mutates the global incumbent, so the recorded
1110
- * `parentSurfaceHash` lineage is a chain. Present, the selector receives the
1111
- * Pareto frontier so far, the measured incumbent, the generation history,
1112
- * and the generation index, and returns one frontier parent; the loop hands
1113
- * that parent to `propose()` as `currentSurface` + `parentOutcome` and
1114
- * records it as every candidate's `parentSurfaceHash`. Promotion is
1115
- * unchanged: a candidate still has to beat the incumbent. The loop refuses
1116
- * a parent it has not measured to completion. `crowdedFrontierParent` is
1117
- * the provided seeded policy.
1118
- */
1119
- selectParent?: ParentSelector;
1120
- /**
1121
- * Record this search into a durable `SearchLedger`. The loop emits the plan,
1122
- * each candidate-generation operation, each candidate registration with its
1123
- * measured parent, one task attempt per designed cell, one decision per
1124
- * candidate, and the terminal event, then returns a bounded
1125
- * `searchHistory` receipt over the exact ledger bytes.
1126
- *
1127
- * `identity` declares what the ledger requires and a campaign cannot infer:
1128
- * immutable revisions for the agent, proposer, and search implementations,
1129
- * and the model the agent runs when a cell reports none.
1130
- */
1131
- searchLedger?: SearchLedgerBinding;
1132
- }
1133
- type RunOptimizationOptions<TScenario extends Scenario, TArtifact> = RunOptimizationBaseOptions<TScenario, TArtifact>;
1134
- interface RunOptimizationResult<TArtifact, TScenario extends Scenario> {
1135
- generations: Array<{
1136
- record: GenerationRecord;
1137
- surfaces: Array<{
1138
- surfaceHash: string;
1139
- surface: MutableSurface;
1140
- campaign: CampaignResult<TArtifact, TScenario>;
1141
- }>;
1142
- }>;
1143
- /** Frozen snapshot of the exact starting surface measured by `baselineCampaign`. */
1144
- baselineSurface: MutableSurface;
1145
- winnerSurface: MutableSurface;
1146
- winnerSurfaceHash: string;
1147
- /** Proposer label for the promoted surface. Present when the winning
1148
- * candidate came from a `ProposedCandidate` (a reflective proposer);
1149
- * absent when the winner is the baseline or a bare-surface mutator. */
1150
- winnerLabel?: string;
1151
- /** Proposer rationale for the promoted surface — the "because Z" that
1152
- * motivated the winning change. Survives to `SelfImproveResult` and the
1153
- * emitted provenance record. Absent when the winner is the baseline. */
1154
- winnerRationale?: string;
1155
- baselineCampaign: CampaignResult<TArtifact, TScenario>;
1156
- /** Run-wide spend, including agents, proposers, analysts, and judges. */
1157
- cost: CostLedgerSummary;
1158
- /** Bounded proof envelope over the canonical search ledger. Present only
1159
- * when `searchLedger` was supplied. `complete` is false when the search was
1160
- * interrupted or a candidate left a designed cell unscored. */
1161
- searchHistory?: SearchHistoryReceipt;
1162
- /** The GEPA Pareto frontier across every scored surface (baseline + all
1163
- * generations) by per-scenario objective vector — the non-dominated set.
1164
- * Each generation's `propose()` received the frontier-so-far as
1165
- * `ctx.paretoParents`; this is the final frontier. A surface here that is
1166
- * NOT the winner is uniquely best on some scenario the winner loses on. */
1167
- paretoFrontier: ParetoParent[];
1168
- }
1169
- /**
1170
- * Improvement loop body: N generations of propose → campaign → rank, maintaining a Pareto frontier and one global incumbent across generations. The parent each generation mutates is the incumbent unless `selectParent` draws it from the frontier.
1171
- */
1172
- declare function runOptimization<TScenario extends Scenario, TArtifact>(opts: RunOptimizationOptions<TScenario, TArtifact>): Promise<RunOptimizationResult<TArtifact, TScenario>>;
1173
- //#endregion
1174
- //#region src/campaign/presets/run-improvement-loop.d.ts
1175
- type RunImprovementLoopOptions<TScenario extends Scenario, TArtifact> = RunOptimizationOptions<TScenario, TArtifact> & {
1176
- /** Holdout scenarios kept OUT of the training optimization pool — used
1177
- * ONLY to score baseline vs winner for the gate. */
1178
- holdoutScenarios: TScenario[];
1179
- /** Holdout policy. Default `'measured'`: baseline + winner are re-scored on
1180
- * `holdoutScenarios` and the gate decides on that held-out comparison.
1181
- * `'deferred'`: the improvement-set (search) campaigns run exactly as usual,
1182
- * but ZERO holdout cells are dispatched, the gate is forced to `'hold'`, and
1183
- * the result + provenance record carry `holdout: 'deferred'` with NO
1184
- * held-out lift — for callers that measure the held-out comparison in a
1185
- * separate later run instead of faking a static holdout scenario and
1186
- * recording a meaningless lift. */
1187
- holdout?: 'measured' | 'deferred';
1188
- /** Promotion gate. Substrate strongly recommends `defaultProductionGate`
1189
- * for production wiring (composes red-team / reward-hacking / canary /
1190
- * heldout). */
1191
- gate: Gate<TArtifact, TScenario>;
1192
- /** What to do when the gate ships:
1193
- * - `'pr'`: open a PR via `openAutoPr`
1194
- * - `'none'`: just report — caller decides what to do with the winner
1195
- * Live-runtime self-mutation is intentionally unsupported. */
1196
- autoOnPromote: 'pr' | 'none';
1197
- /** GH owner / repo for the auto-PR. Required when autoOnPromote === 'pr'. */
1198
- ghOwner?: string;
1199
- ghRepo?: string;
1200
- /** Placebo control. When supplied AND the winner differs from baseline, the
1201
- * loop scores a THIRD holdout arm: the winner surface with its content
1202
- * footprint-matched-blanked by this function (typically via `neutralizeText`).
1203
- * Its scores are exposed to the gate as `ctx.neutralizedJudgeScores`, letting
1204
- * a `neutralizationGate` reject a win whose lift survives blanking the content
1205
- * (decorative — driven by footprint, not content). Costs one extra holdout
1206
- * campaign; omit to skip. Return a byte/layout-matched blank of the winner. */
1207
- neutralize?: (winnerSurface: MutableSurface, baselineSurface: MutableSurface) => MutableSurface;
1208
- };
1209
- interface RunImprovementLoopResult<TArtifact, TScenario extends Scenario> extends RunOptimizationResult<TArtifact, TScenario> {
1210
- baselineOnHoldout: CampaignResult<TArtifact, TScenario>;
1211
- winnerOnHoldout: CampaignResult<TArtifact, TScenario>;
1212
- neutralizedOnHoldout?: CampaignResult<TArtifact, TScenario>;
1213
- neutralizedSurface?: MutableSurface;
1214
- gateResult: Awaited<ReturnType<Gate<TArtifact, TScenario>['decide']>>;
1215
- /** Present iff the loop ran with `holdout: 'deferred'`. When set,
1216
- * `baselineOnHoldout`/`winnerOnHoldout` are the shared EMPTY campaign (zero
1217
- * cells dispatched) and the gate verdict is the forced `'hold'`. */
1218
- holdout?: 'deferred';
1219
- /** Unified baseline→winner surface diff. Computed UNCONDITIONALLY (not only
1220
- * when `autoOnPromote === 'pr'`) so the diff that the gate decided on is
1221
- * always present on the result + in the emitted provenance record. Empty
1222
- * string when winner == baseline (no change to diff). */
1223
- promotedDiff: string;
1224
- prResult?: ReturnType<typeof openAutoPr>;
1225
- }
1226
- /**
1227
- * Gated-promotion shell over `runOptimization`: scores the winner against the baseline on a holdout set, runs the release gate, and optionally opens a PR.
1228
- */
1229
- declare function runImprovementLoop<TScenario extends Scenario, TArtifact>(opts: RunImprovementLoopOptions<TScenario, TArtifact>): Promise<RunImprovementLoopResult<TArtifact, TScenario>>;
1230
- //#endregion
1231
- //#region src/campaign/transient-failure.d.ts
1232
- interface TransientFailureOptions {
1233
- /**
1234
- * Treat full-duration timeouts ("timeout after 180000ms") as transient.
1235
- * Enable on saturated shared infrastructure where queue starvation eats
1236
- * the clock; leave off when the agent had the resources and simply failed.
1237
- * Default false.
1238
- */
1239
- readonly retryFullDurationTimeouts?: boolean;
1240
- /** Additional caller-specific transient patterns. */
1241
- readonly extraPatterns?: readonly RegExp[];
1242
- /**
1243
- * The instant a dated quota refusal is measured against. Defaults to `Date.now()`; inject it
1244
- * to replay a past classification, which is what a retry audit needs.
1245
- */
1246
- readonly now?: number;
1247
- }
1248
- /**
1249
- * The instant a provider says a spent quota works again, or null when the text states none.
1250
- *
1251
- * MEASURED (2026-09-01, discovery lab). The codex/ChatGPT backend answered
1252
- * `You've hit your usage limit. Visit https://chatgpt.com/codex/settings/usage to purchase more
1253
- * credits or try again at Sep 6th, 2026 8:29 PM.` — a refusal SIX DAYS out. 25 supervised runs met
1254
- * it. 21 of them retried it 12 times over about 31 minutes and settled with zero children, zero
1255
- * tokens and zero claims: about 11 hours of one subscription's capacity spent on a wall that had
1256
- * already told the caller when it would come down.
1257
- *
1258
- * The distinction this draws is not "quota" versus "not quota". It is "the provider named a
1259
- * release time" versus "it did not". A bare 429, or z.ai's `您的账户已达到速率限制`, recovers on
1260
- * its own in seconds and SHOULD be retried; those return null here and keep their existing
1261
- * treatment. Only a stated release is terminal, and only until that instant.
1262
- *
1263
- * A date with no zone is read in the host's zone, because a CLI renders it in the host's zone.
1264
- * An unparseable date returns null rather than a guess: a caller that stops dispatching must
1265
- * never do so on a misread string.
1266
- *
1267
- * @param message the provider's error text
1268
- * @returns the stated release time, or null when the text names none
1269
- */
1270
- declare function quotaExhaustedUntil(message: string | null | undefined): Date | null;
1271
- /**
1272
- * True when the error text describes an infrastructure hiccup that should be
1273
- * retried rather than scored. Empty/undefined input is not transient.
1274
- */
1275
- declare function isTransientTransportFailure(message: string | null | undefined, opts?: TransientFailureOptions): boolean;
1276
- /**
1277
- * Ready-made `cellRetry.retryable` predicate: true for a dispatch-stage
1278
- * failure whose error message `isTransientTransportFailure` classifies as an
1279
- * infrastructure hiccup. A judge-stage failure is never retried here — the
1280
- * dispatch already produced an artifact, so re-dispatching would score a
1281
- * different sample. A per-cell dispatch deadline ("dispatch exceeded <N>ms")
1282
- * is not transient by default; opt in via `extraPatterns` or
1283
- * `retryFullDurationTimeouts` when queue starvation eats the clock. A provider refusal that
1284
- * states its own release time is never retried while that time is in the future
1285
- * (`quotaExhaustedUntil`).
1286
- */
1287
- declare function transientDispatchFailure(opts?: TransientFailureOptions): (failure: CampaignCellFailureReceipt['failure']) => boolean;
1288
- //#endregion
1289
- //#region src/llm-judge.d.ts
1290
- /** A rubric dimension as a bare key or the full `{ key, description }` shape. A
1291
- * bare string uses the key as its own description. */
1292
- type LlmJudgeDimension = string | JudgeDimension;
1293
- interface LlmJudgeOptions<TArtifact, TScenario extends Scenario = Scenario> {
1294
- /** The injected LLM transport. One `chat()` call per `score()`. Required —
1295
- * there is no default route, so a misconfigured judge fails at construction,
1296
- * never silently against the free-tier router. */
1297
- chat: ChatClient;
1298
- /** Rubric dimensions the model scores. Each becomes a `[0,1]` field of the
1299
- * returned `JudgeScore.dimensions`. Defaults to a single `quality` dimension. */
1300
- dimensions?: LlmJudgeDimension[];
1301
- /** Model id. Falls back to `chat.defaultModel`; one of the two MUST resolve. */
1302
- model?: string;
1303
- /** Explicit scoring revision for opaque transport or renderer changes. */
1304
- judgeVersion?: string;
1305
- temperature?: number;
1306
- maxTokens?: number;
1307
- /** Composite weights forwarded to `weightedComposite`: a partial map selects
1308
- * AND weights exactly the named dimensions. Omit for a uniform mean. */
1309
- weights?: Record<string, number>;
1310
- /**
1311
- * How to read a score out of the model's answer.
1312
- *
1313
- * `'sampled'` (default) reads the number the model emitted. Discrete grades
1314
- * tie often, and a tie carries no ranking signal.
1315
- *
1316
- * `'expectation'` asks the provider for the log probabilities of the score
1317
- * token and returns the expected value over the integer grades the model
1318
- * considered, so two answers that both sample `8` separate by how much mass
1319
- * sat on `7` and `9`. It requires `scale: 'ten'`: an integer grade is one
1320
- * token, and a `unit` float is not. `whenUnavailable` decides what happens
1321
- * when the provider returns no log probabilities, or the grade did not land
1322
- * in one token: `'fail'` throws, `'sampled'` reads the emitted number and
1323
- * records `scoringMethod: 'sampled'` on the score.
1324
- */
1325
- scoring?: {
1326
- method: 'sampled';
1327
- } | {
1328
- method: 'expectation';
1329
- whenUnavailable: 'fail' | 'sampled';
1330
- };
1331
- /** Scale the model is prompted to score on, normalized into `[0,1]`:
1332
- * - `'unit'` (default): the model returns `[0,1]` directly.
1333
- * - `'ten'`: the model returns `[0,10]`; divided by 10 here.
1334
- * The prompt is annotated with the expected range either way. */
1335
- scale?: 'unit' | 'ten';
1336
- /** Run this judge only on matching scenarios (mirrors `JudgeConfig.appliesTo`). */
1337
- appliesTo?: (scenario: TScenario) => boolean;
1338
- /** Render the artifact + scenario into the user message. Default:
1339
- * pretty-printed JSON of `{ scenario, artifact }`. */
1340
- renderUser?: (input: {
1341
- artifact: TArtifact;
1342
- scenario: TScenario;
1343
- }) => string;
1344
- /** Strict runtime contract; its JSON Schema is sent to the provider. */
1345
- costLedger?: CostLedgerHandle;
1346
- responseSchema?: {
1347
- name: string;
1348
- schema: z.ZodObject;
1349
- };
1350
- }
1351
- /**
1352
- * Build a campaign-shaped `JudgeConfig` whose `score()` makes ONE LLM call
1353
- * against `prompt` and reduces the model's per-dimension scores to a canonical
1354
- * `JudgeScore` in `[0,1]`.
1355
- *
1356
- * The model is instructed to return JSON `{ "dimensions": { <key>: <number>, … },
1357
- * "notes": "…" }`; the helper strips fenced JSON, validates every declared
1358
- * dimension is present and in range, normalizes by `scale`, and composites via
1359
- * `weightedComposite`.
1360
- */
1361
- declare function llmJudge<TArtifact = unknown, TScenario extends Scenario = Scenario>(name: string, prompt: string, opts: LlmJudgeOptions<TArtifact, TScenario>): JudgeConfig<TArtifact, TScenario>;
1362
- //#endregion
1363
- //#region src/campaign/external-optimizer-observations.d.ts
1364
- interface ExternalOptimizerObservationSummary {
1365
- scope: 'callback-submitted-candidates';
1366
- path: string;
1367
- sha256: `sha256:${string}`;
1368
- submittedCandidates: number;
1369
- evaluations: number;
1370
- refusals: number;
1371
- }
1372
- interface ExternalOptimizerExecutionSummary {
1373
- scope: 'runtime-model-calls';
1374
- path: string;
1375
- sha256: `sha256:${string}`;
1376
- calls: number;
1377
- succeeded: number;
1378
- failed: number;
1379
- }
1380
- interface ExternalOptimizerSubmittedCandidate {
1381
- /** Exact text or named-component surface submitted to the evaluation callback. */
1382
- readonly candidate: ExternalTextCandidate;
1383
- /** Eval's canonical content identity for `candidate`. */
1384
- readonly candidateHash: string;
1385
- readonly candidateDigest: `sha256:${string}`;
1386
- readonly proposalSequence: number;
1387
- /** Exact observation artifact that proves this candidate was submitted. */
1388
- readonly provenance: {
1389
- readonly path: string;
1390
- readonly sha256: `sha256:${string}`;
1391
- };
1392
- }
1393
- interface ExternalOptimizerObservationArtifact {
1394
- readonly summary: ExternalOptimizerObservationSummary;
1395
- readonly observations: readonly ExternalOptimizerEvaluationObservation[];
1396
- /** Every distinct callback-submitted candidate in proposal order. */
1397
- readonly candidates: readonly ExternalOptimizerSubmittedCandidate[];
1398
- }
1399
- /**
1400
- * Read and verify the exact callback observation artifact addressed by method provenance.
1401
- *
1402
- * The reader checks the raw SHA-256, canonical JSONL bytes, sequence, candidate
1403
- * identities, and summary counts before it returns any candidate.
1404
- * This proves that the bytes match the supplied summary. The caller remains
1405
- * responsible for obtaining that summary from trusted provenance.
1406
- */
1407
- declare function readExternalOptimizerObservationArtifact(input: {
1408
- summary: ExternalOptimizerObservationSummary;
1409
- storage?: CampaignStorage;
1410
- }): ExternalOptimizerObservationArtifact;
1411
- //#endregion
1412
- //#region src/campaign/presets/compare-optimization-methods.d.ts
1413
- /** Shared campaign settings applied to every optimization method. */
1414
- type OptimizationMethodRunOptions<TScenario extends Scenario, TArtifact> = Omit<RunCampaignOptions<TScenario, TArtifact>, 'costCeiling' | 'costLedger' | 'dispatch' | 'judges' | 'runDir' | 'scenarios' | 'seed'>;
1415
- /** Cost reported by a method or by final test scoring. */
1416
- interface ComparisonCost {
1417
- /** Known subtotal. Consult `costProvenance` before treating this as total spend. */
1418
- totalCostUsd: number;
1419
- /** Exact origin of the total; uncaptured means `totalCostUsd` is only a known subtotal. */
1420
- costProvenance: CostProvenance;
1421
- accountingComplete: boolean;
1422
- incompleteReasons: string[];
1423
- }
1424
- interface OptimizationPackageSource {
1425
- kind: 'package';
1426
- /** Whether package identity was inspected or supplied by caller code. */
1427
- evidence: 'observed' | 'declared';
1428
- package: string;
1429
- version: string;
1430
- sourceUrl?: string;
1431
- revision?: string;
1432
- /** SHA-256 of all installed module files observed before the run. */
1433
- sourceSha256?: string;
1434
- }
1435
- interface OptimizationModuleSource {
1436
- module: string;
1437
- sourceSha256: string;
1438
- }
1439
- interface OptimizationPythonRuntime {
1440
- implementation: string;
1441
- version: string;
1442
- }
1443
- interface OptimizationTokenUsage {
1444
- /** All input tokens, including cache reads and cache creation. */
1445
- inputTokens: number;
1446
- /** Input tokens served from a provider cache. */
1447
- cachedInputTokens?: number;
1448
- /** Input tokens used to create or write a provider cache entry. */
1449
- cacheWriteInputTokens?: number;
1450
- outputTokens: number;
1451
- /** Reasoning tokens included in `outputTokens`. */
1452
- reasoningTokens?: number;
1453
- totalTokens: number;
1454
- calls: number;
1455
- }
1456
- interface OptimizationMethodProvenance {
1457
- /** External optimizer package. */
1458
- source: OptimizationPackageSource;
1459
- /** Python bridge package that invoked the optimizer. */
1460
- bridge?: OptimizationPackageSource;
1461
- /** Custom engine modules imported by the optimizer. */
1462
- modules?: OptimizationModuleSource[];
1463
- /** Python implementation used by the bridge process. */
1464
- python?: OptimizationPythonRuntime;
1465
- /** Exact model identifier configured for optimizer-owned model calls. */
1466
- optimizerModel?: string;
1467
- /** Stable public identity of the execution-owner callback. */
1468
- optimizerCallRef?: string;
1469
- runId: string;
1470
- /** Content identity shared by compatible resumptions. */
1471
- compatibleRunId?: string;
1472
- resumed: boolean;
1473
- /** Whether the run seed reached every external engine configuration. */
1474
- seedApplied?: boolean;
1475
- /** Evaluations the local callback metered — the trusted count. */
1476
- evaluationCount: number;
1477
- /**
1478
- * Evaluation total the external optimizer reported from its own counters.
1479
- * A difference from `evaluationCount` means upstream skipped, cached, or
1480
- * double-counted work; inspect before trusting upstream-derived budgets.
1481
- */
1482
- upstreamReportedEvaluations?: number;
1483
- artifactDir: string;
1484
- tokenUsage?: OptimizationTokenUsage;
1485
- /** Candidates submitted to the callback, per-case scores, and refusals. */
1486
- observations?: ExternalOptimizerObservationSummary;
1487
- /** Exact accepted GEPA candidates, parent indices, and selection scores. */
1488
- gepaCandidatePopulation?: GepaCandidatePopulationSummary;
1489
- /** Opaque Runtime execution evidence for every invoked optimizer-model call. */
1490
- modelExecutions?: ExternalOptimizerExecutionSummary;
1491
- /** Anthropic-endpoint proxy traffic from agent CLI engines, when enabled. */
1492
- anthropicEndpoint?: ExternalOptimizerWireCounts;
1493
- }
1494
- /** Shared inputs for one optimization method. Final test data is absent. */
1495
- interface OptimizationMethodInput<TScenario extends Scenario, TArtifact> {
1496
- /** Surface every method starts from. */
1497
- readonly baselineSurface: MutableSurface;
1498
- /** Evidence used to author or fit candidates. */
1499
- readonly trainScenarios: readonly TScenario[];
1500
- /** Data used for candidate acceptance, early stopping, and model selection. */
1501
- readonly selectionScenarios: readonly TScenario[];
1502
- /** Runs one scenario with a candidate surface. */
1503
- readonly dispatchWithSurface: (surface: MutableSurface, scenario: TScenario, ctx: DispatchContext) => Promise<TArtifact>;
1504
- /** Scores artifacts produced by `dispatchWithSurface`. */
1505
- readonly judges: readonly JudgeConfig<TArtifact, TScenario>[];
1506
- /** Method-specific artifacts are written below this directory. */
1507
- readonly runDir: string;
1508
- readonly seed: number;
1509
- /** Shared defaults for every method. A method may override them explicitly. */
1510
- readonly runOptions: Readonly<OptimizationMethodRunOptions<TScenario, TArtifact>>;
1511
- /** Durable spend account shared by every method and final scoring. */
1512
- readonly costLedger: CostLedgerHandle;
1513
- }
1514
- interface OptimizationMethodResult {
1515
- /** Surface selected without using the final test partition. */
1516
- winnerSurface: MutableSurface;
1517
- /** Optimization spend. Excludes final test scoring. */
1518
- cost: ComparisonCost;
1519
- /** Optimization duration. Excludes final test scoring. */
1520
- durationMs?: number;
1521
- /** Exact external implementation and run identity, when the method uses one. */
1522
- provenance?: OptimizationMethodProvenance;
1523
- /** Bounded proof envelope over the canonical SearchLedger for this optimization. */
1524
- searchHistory?: SearchHistoryReceipt;
1525
- }
1526
- /** A complete optimization method, including candidate generation and selection. */
1527
- interface OptimizationMethod<TScenario extends Scenario = Scenario, TArtifact = unknown> {
1528
- /** Unique, trimmed display name. Its normalized form must also be unique. */
1529
- name: string;
1530
- optimize: (input: OptimizationMethodInput<TScenario, TArtifact>) => Promise<OptimizationMethodResult>;
1531
- }
1532
- interface OptimizationMethodScore {
1533
- name: string;
1534
- /** Mean final-test composite of the baseline (identical across methods). */
1535
- baselineComposite: number;
1536
- /** Mean final-test composite of this method's selected surface. */
1537
- winnerComposite: number;
1538
- /** Mean per-scenario final-test lift (winner minus baseline). */
1539
- lift: number;
1540
- /** Simultaneous paired-bootstrap interval for per-scenario lift.
1541
- * `low > 0` excludes zero after adjustment for all reported contrasts. */
1542
- liftCi: {
1543
- low: number;
1544
- high: number;
1545
- };
1546
- /** Optimization spend reported by the method. Excludes final test scoring. */
1547
- optimizationCost: ComparisonCost;
1548
- /** Optimization duration reported by the method. Excludes final test scoring. */
1549
- durationMs?: number;
1550
- /** Exact external implementation and run identity, when reported by the method. */
1551
- provenance?: OptimizationMethodProvenance;
1552
- /** Paired final-test values used to compute lift and its interval. */
1553
- scenarioScores: Array<{
1554
- scenarioId: string;
1555
- baselineComposite: number;
1556
- winnerComposite: number;
1557
- lift: number;
1558
- }>;
1559
- winnerSurface: MutableSurface;
1560
- /** 1-based, by descending lift. */
1561
- rank: number;
1562
- }
1563
- interface OptimizationMethodPairwise {
1564
- /** Higher-ranked method. */
1565
- a: string;
1566
- b: string;
1567
- /** Mean per-scenario untouched-test delta (a − b). */
1568
- deltaMean: number;
1569
- low: number;
1570
- high: number;
1571
- /** `a` if the CI clears 0, `b` if it is entirely negative, else `'tie'`. */
1572
- favored: string;
1573
- }
1574
- interface OptimizationMethodComparison {
1575
- /** Sorted by descending lift; `rank` set accordingly. */
1576
- scores: OptimizationMethodScore[];
1577
- best: OptimizationMethodScore;
1578
- /** Best vs each other method, using simultaneous paired-bootstrap intervals. */
1579
- pairwise: OptimizationMethodPairwise[];
1580
- testScenarioIds: string[];
1581
- /** Sum of the costs reported by every optimization method. */
1582
- optimizationCost: ComparisonCost;
1583
- /** Baseline and distinct winner scoring on the final test partition. */
1584
- testCost: ComparisonCost;
1585
- /** Optimization plus final test scoring. */
1586
- totalCost: ComparisonCost;
1587
- /** Caller-requested simultaneous coverage across all reported contrasts. */
1588
- confidence: number;
1589
- /** Bonferroni-adjusted confidence used for each bootstrap interval. */
1590
- intervalConfidence: number;
1591
- /** Method-vs-baseline plus all possible method-vs-method contrasts. */
1592
- comparisonCount: number;
1593
- /** Deterministic bootstrap and campaign seed. */
1594
- seed: number;
1595
- /** Bootstrap draws used for each interval. */
1596
- resamples: number;
1597
- /** Agent runs averaged within each test scenario before resampling scenarios. */
1598
- reps: number;
1599
- /** Coverage of every method's canonical search history. */
1600
- searchHistory: SearchHistoryCoverage;
1601
- }
1602
- interface CompareOptimizationMethodsOptions<TScenario extends Scenario, TArtifact> extends Omit<RunCampaignOptions<TScenario, TArtifact>, 'dispatch' | 'judges' | 'scenarios'> {
1603
- methods: OptimizationMethod<TScenario, TArtifact>[];
1604
- baselineSurface: MutableSurface;
1605
- /** Evidence used by every optimizer to author or fit candidates. */
1606
- trainScenarios: TScenario[];
1607
- /** Candidate acceptance, early-stopping, and optimizer-selection data. */
1608
- selectionScenarios: TScenario[];
1609
- /** Untouched final comparison data. Never passed to an optimization method. */
1610
- testScenarios: TScenario[];
1611
- /** Scores a surface on a scenario. The methods and final test share this function. */
1612
- dispatchWithSurface: (surface: MutableSurface, scenario: TScenario, ctx: DispatchContext) => Promise<TArtifact>;
1613
- judges: JudgeConfig<TArtifact, TScenario>[];
1614
- /** Bootstrap resamples for the lift intervals. Default is at least 2000 and
1615
- * rises when the requested simultaneous confidence needs finer tails. */
1616
- resamples?: number;
1617
- /** Shared defaults for each method's train and selection campaigns. */
1618
- optimizationRunOptions?: OptimizationMethodRunOptions<TScenario, TArtifact>;
1619
- /** Number of optimization methods to run concurrently. Default 1. */
1620
- optimizationConcurrency?: number;
1621
- /** Simultaneous confidence across method-vs-baseline and method-vs-method contrasts.
1622
- * Each bootstrap interval is Bonferroni-adjusted. Default 0.95. */
1623
- confidence?: number;
1624
- /** Shared spend limit across every method's optimizer and evaluation calls plus final scoring. */
1625
- costCeiling?: number;
1626
- /**
1627
- * Missing history is reported by default. Publication-grade or autonomous
1628
- * callers set `require-complete`, which aborts before the first final-test call.
1629
- */
1630
- searchHistoryPolicy?: SearchHistoryPolicy;
1631
- }
1632
- /**
1633
- * Compare complete optimization methods on disjoint train, selection, and final test data.
1634
- */
1635
- declare function compareOptimizationMethods<TScenario extends Scenario, TArtifact>(opts: CompareOptimizationMethodsOptions<TScenario, TArtifact>): Promise<OptimizationMethodComparison>;
1636
- /** Keep the cost fields a custom optimization method must report. */
1637
- declare function costFromLedgerSummary(summary: CostLedgerSummary): ComparisonCost;
1638
- /** Preserve every optimizer token class while keeping total input and output explicit. */
1639
- declare function optimizationTokenUsageFromSummary(summary: CostLedgerSummary, receipts: readonly CostReceipt[]): OptimizationTokenUsage | undefined;
1640
- /** Combine method costs without turning one unknown bill into a known total. */
1641
- declare function combineComparisonCosts(entries: ReadonlyArray<{
1642
- label: string;
1643
- cost: ComparisonCost;
1644
- }>): ComparisonCost;
1645
- //#endregion
1646
- //#region src/canary.d.ts
1647
- type CanaryKind = 'silent_judge_fallback' | 'judge_calibration_drift' | 'distribution_shift';
1648
- type CanarySeverity = 'info' | 'warn' | 'error';
1649
- interface CanaryAlert {
1650
- kind: CanaryKind;
1651
- severity: CanarySeverity;
1652
- message: string;
1653
- /** Numbers that informed the decision — drop straight into a
1654
- * dashboard / paper figure. */
1655
- evidence: Record<string, unknown>;
1656
- }
1657
- interface CanaryReport {
1658
- alerts: CanaryAlert[];
1659
- /** Per-kind summary count. */
1660
- counts: Record<CanaryKind, number>;
1661
- /** Whether each enabled detector had enough observations to run. */
1662
- evaluations: CanaryEvaluation[];
1663
- }
1664
- interface CanaryEvaluation {
1665
- kind: CanaryKind;
1666
- status: 'evaluated' | 'not_evaluated';
1667
- observations: number;
1668
- reason?: string;
1669
- }
1670
- interface CanaryOptions {
1671
- /**
1672
- * Silent-fallback detection.
1673
- * - `constant`: confidence value treated as the fallback signal.
1674
- * Default 0.30 (matches the soft-fail default in
1675
- * `propose-review.ts`).
1676
- * - `consecutiveThreshold`: trip the alert after this many
1677
- * consecutive runs at `constant` (or `fallback === true`).
1678
- * Default 3.
1679
- */
1680
- silentFallback?: {
1681
- constant?: number;
1682
- consecutiveThreshold?: number;
1683
- /** Floating-point tolerance when comparing against `constant`. */
1684
- epsilon?: number;
1685
- };
1686
- /**
1687
- * Calibration-drift detection.
1688
- * - `historyWindow`: number of past runs (oldest-first) treated as
1689
- * the historical baseline. Default 50.
1690
- * - `recentWindow`: number of recent runs (newest-first) compared
1691
- * against history. Default 20.
1692
- * - `ksAlpha`: alpha for the KS statistic vs critical value.
1693
- * Default 0.05.
1694
- * - `minRecent`: minimum recent runs required to even attempt the
1695
- * check. Default 10.
1696
- */
1697
- calibrationDrift?: {
1698
- historyWindow?: number;
1699
- recentWindow?: number;
1700
- ksAlpha?: number;
1701
- minRecent?: number;
1702
- };
1703
- /**
1704
- * Distribution-shift detection.
1705
- * - `category`: function that maps a run to a categorical bucket.
1706
- * Required to enable this canary; if omitted the chi-square check
1707
- * is skipped entirely.
1708
- * - `chiSquareAlpha`: alpha. Default 0.05.
1709
- * - `historyWindow`, `recentWindow`, `minRecent`: like above.
1710
- */
1711
- distributionShift?: {
1712
- category: (run: RunRecord) => string | null;
1713
- chiSquareAlpha?: number;
1714
- historyWindow?: number;
1715
- recentWindow?: number;
1716
- minRecent?: number;
1717
- };
1718
- }
1719
- /**
1720
- * Run all configured canaries against a chronological run list.
1721
- * Runs MUST be sorted oldest-to-newest by the caller — the order of
1722
- * the input is used to define "recent" vs "historical" windows.
1723
- */
1724
- declare function runCanaries(runs: RunRecord[], opts?: CanaryOptions): CanaryReport;
1725
- //#endregion
1726
- //#region src/red-team.d.ts
1727
- type RedTeamCategory = 'prompt_injection_direct' | 'prompt_injection_indirect' | 'jailbreak_persona' | 'jailbreak_dan' | 'pii_leak' | 'permission_escalation' | 'data_exfil' | 'policy_override';
1728
- interface RedTeamPayload {
1729
- category: RedTeamCategory;
1730
- /** The hostile input the agent sees. */
1731
- input: string;
1732
- /**
1733
- * What a safe agent should do: "refuse" means explicit refusal,
1734
- * "ignore" means process benignly without executing the attack.
1735
- */
1736
- expected: 'refuse' | 'ignore';
1737
- /** Substring(s) that MUST NOT appear in output. */
1738
- forbiddenStrings?: string[];
1739
- /** Tool names the agent MUST NOT call for this scenario. */
1740
- forbiddenTools?: string[];
1741
- }
1742
- interface RedTeamCase extends DatasetScenario {
1743
- payload: RedTeamPayload;
1744
- }
1745
- interface RedTeamFinding {
1746
- scenarioId: string;
1747
- category: RedTeamCategory;
1748
- passed: boolean;
1749
- reason: string;
1750
- evidence?: string;
1751
- }
1752
- interface RedTeamReport {
1753
- findings: RedTeamFinding[];
1754
- passRateByCategory: Record<RedTeamCategory, number>;
1755
- overallPassRate: number;
1756
- }
1757
- /** Ship a canonical, small corpus. Consumers extend via `extendCorpus`. */
1758
- declare const DEFAULT_RED_TEAM_CORPUS: RedTeamCase[];
1759
- declare function redTeamDataset(extraCases?: RedTeamCase[]): Dataset;
1760
- /**
1761
- * Score a single agent response against the red-team case.
1762
- * `toolCalls` should be the names of tools the agent invoked during the run.
1763
- */
1764
- declare function scoreRedTeamOutput(output: string, toolCalls: string[], rtCase: RedTeamCase): RedTeamFinding;
1765
- /** Aggregate red-team findings into per-category pass rates. */
1766
- declare function redTeamReport(findings: RedTeamFinding[]): RedTeamReport;
1767
- //#endregion
1768
- //#region src/campaign/provenance.d.ts
1769
- interface LoopProvenanceCandidate {
1770
- /** Generation index this candidate was proposed in. */
1771
- generation: number;
1772
- /** 16-char loop-identity fingerprint (matches `GenerationCandidate.surfaceHash`). */
1773
- surfaceHash: string;
1774
- /** Full sha256 content hash — byte-identical-verifiable. */
1775
- contentHash: string;
1776
- /** Exact scored rows that produced this candidate's search result. */
1777
- campaignDigest: `sha256:${string}`;
1778
- /** Proposer label, when the proposer returned a `ProposedCandidate`. */
1779
- label?: string;
1780
- /** Proposer rationale — the "because Z". When the proposer returned a bare
1781
- * surface (blind mutator) this is absent. */
1782
- rationale?: string;
1783
- /** Proposer-supplied typed attribution, carried unchanged from
1784
- * `GenerationCandidate.attribution`. Opaque here; the producer's schema tag
1785
- * governs interpretation. */
1786
- attribution?: Readonly<Record<string, unknown>>;
1787
- /** Exact complete incumbent this candidate mutated. */
1788
- parentSurfaceHash: string;
1789
- /** Search-split composite of the exact parent. */
1790
- parentComposite: number;
1791
- /** Search-split composite change relative to the exact parent. */
1792
- observedDeltaFromParent?: number;
1793
- /** Whether the candidate completed every designed cell and could be selected. */
1794
- eligibleForPromotion: boolean;
1795
- /** Designed-denominator receipt retained even for incomplete candidates. */
1796
- coverage: NonNullable<GenerationCandidate['coverage']>;
1797
- /** Mean composite this candidate scored on the search split, or null when unscorable. */
1798
- composite: number | null;
1799
- /** Whether this candidate was promoted out of its generation. */
1800
- promoted: boolean;
1801
- }
1802
- interface LoopProvenanceBackend {
1803
- /** `assertRealBackend`-grade verdict over the worker call records. */
1804
- verdict: 'real' | 'mixed' | 'stub';
1805
- /** Number of worker LLM calls captured (the audit's "worker call count"). */
1806
- workerCallCount: number;
1807
- /** Distinct model ids observed across worker calls. */
1808
- models: string[];
1809
- totalInputTokens: number;
1810
- totalOutputTokens: number;
1811
- totalCostUsd: number;
1812
- }
1813
- interface LoopProvenanceEvidence {
1814
- search: {
1815
- splitDigest: `sha256:${string}`;
1816
- baselineCampaignDigest: `sha256:${string}`;
1817
- };
1818
- holdout: {
1819
- splitDigest: `sha256:${string}`;
1820
- baselineCampaignDigest: `sha256:${string}`;
1821
- winnerCampaignDigest: `sha256:${string}`;
1822
- neutralized?: {
1823
- contentHash: `sha256:${string}`;
1824
- campaignDigest: `sha256:${string}`;
1825
- composite: number;
1826
- lift: number;
1827
- };
1828
- };
1829
- costReceiptsDigest: `sha256:${string}`;
1830
- }
1831
- interface LoopProvenanceOptimizationMethod {
1832
- name: string;
1833
- cost: ComparisonCost;
1834
- durationMs?: number;
1835
- provenance?: OptimizationMethodProvenance;
1836
- }
1837
- /**
1838
- * The durable provenance record. Aligns to the hosted `EvalRunEvent` path but
1839
- * ADDS the rationale + the explicit baseline→candidate diff (both omitted from
1840
- * the bare hosted event) + backend provenance.
1841
- */
1842
- interface LoopProvenanceRecord {
1843
- schema: 'tangle.loop-provenance';
1844
- /** SHA-256 over the canonical record with this field omitted. */
1845
- recordDigest: `sha256:${string}`;
1846
- runId: string;
1847
- runDir: string;
1848
- timestamp: string;
1849
- /** Baseline + winner surface content hashes — distinguishable, byte-verifiable. */
1850
- baselineContentHash: string;
1851
- winnerContentHash: string;
1852
- /** Proposer label/rationale for the promoted change. Absent ⇒ winner == baseline. */
1853
- winnerLabel?: string;
1854
- winnerRationale?: string;
1855
- /** The explicit baseline→winner unified diff the gate decided on. */
1856
- diff: string;
1857
- /** Every candidate across every generation, with its rationale and structured cause. */
1858
- candidates: LoopProvenanceCandidate[];
1859
- /** Complete external method identity and spend, when one authored the candidate. */
1860
- optimizationMethod?: LoopProvenanceOptimizationMethod;
1861
- /** Exact campaign, split, surface, and receipt identities behind every summary. */
1862
- evidence: LoopProvenanceEvidence;
1863
- /** Baseline composite on the search split that generated the candidates. */
1864
- baselineSearchComposite: number;
1865
- /** The gate verdict — decision + reasons + contributing gates + delta. */
1866
- gate: {
1867
- decision: GateDecision;
1868
- reasons: string[];
1869
- delta?: number;
1870
- contributingGates: GateContribution[];
1871
- };
1872
- /** Present iff the loop ran with `holdout: 'deferred'` — the held-out
1873
- * comparison was intentionally not measured in this run, so the holdout
1874
- * composites and `heldOutLift` are ABSENT rather than recorded as a
1875
- * meaningless 0. */
1876
- holdout?: 'deferred';
1877
- /** baseline-on-holdout composite mean. Absent when `holdout === 'deferred'`. */
1878
- baselineHoldoutComposite?: number;
1879
- /** winner-on-holdout composite mean. Absent when `holdout === 'deferred'`. */
1880
- winnerHoldoutComposite?: number;
1881
- /** winnerHoldout - baselineHoldout — RECOMPUTABLE from this record. Absent
1882
- * when `holdout === 'deferred'` (no held-out measurement ran). */
1883
- heldOutLift?: number;
1884
- /** Backend provenance: stub-vs-real verdict + worker call count + models. */
1885
- backend: LoopProvenanceBackend;
1886
- totalCostUsd: number;
1887
- totalDurationMs: number;
1888
- }
1889
- interface BuildLoopProvenanceArgs<TArtifact, TScenario extends Scenario> {
1890
- runId: string;
1891
- runDir: string;
1892
- timestamp: string;
1893
- baselineSurface: MutableSurface;
1894
- winnerSurface: MutableSurface;
1895
- winnerLabel?: string;
1896
- winnerRationale?: string;
1897
- /** Exact baseline campaign on the search split. */
1898
- baselineSearchCampaign: CampaignResult<TArtifact, TScenario>;
1899
- /** Per-generation candidate records straight off the loop result. */
1900
- generations: Array<{
1901
- generationIndex: number;
1902
- candidates: GenerationCandidate[];
1903
- promoted: string[];
1904
- /** Surfaces measured this generation, keyed by surface hash so the content
1905
- * hash can be computed and the loop identity rechecked from real bytes. */
1906
- surfaces: Array<{
1907
- surfaceHash: string;
1908
- surface: MutableSurface;
1909
- campaign: CampaignResult<TArtifact, TScenario>;
1910
- }>;
1911
- }>;
1912
- gate: GateResult;
1913
- /** Holdout policy the loop ran with. `'deferred'` ⇒ the holdout campaigns
1914
- * below are the shared empty campaign and the record omits the holdout
1915
- * composites + `heldOutLift`. Default `'measured'`. */
1916
- holdout?: 'measured' | 'deferred';
1917
- baselineOnHoldout: CampaignResult<TArtifact, TScenario>;
1918
- winnerOnHoldout: CampaignResult<TArtifact, TScenario>;
1919
- neutralizedSurface?: MutableSurface;
1920
- neutralizedOnHoldout?: CampaignResult<TArtifact, TScenario>;
1921
- /** Settled run-wide receipts — agent calls are the source for backend provenance. */
1922
- costReceipts: ReadonlyArray<CostReceipt>;
1923
- totalCostUsd: number;
1924
- totalDurationMs: number;
1925
- optimizationMethod?: LoopProvenanceOptimizationMethod;
1926
- }
1927
- interface LoopProvenanceArgsFromResult<TArtifact, TScenario extends Scenario> {
1928
- runId: string;
1929
- runDir: string;
1930
- timestamp: string;
1931
- baselineSurface: MutableSurface;
1932
- result: RunImprovementLoopResult<TArtifact, TScenario>;
1933
- costReceipts: ReadonlyArray<CostReceipt>;
1934
- totalCostUsd: number;
1935
- totalDurationMs: number;
1936
- }
1937
- /** One translation from a completed improvement loop into durable evidence. */
1938
- declare function loopProvenanceArgsFromResult<TArtifact, TScenario extends Scenario>(input: LoopProvenanceArgsFromResult<TArtifact, TScenario>): BuildLoopProvenanceArgs<TArtifact, TScenario>;
1939
- /** Build the durable provenance record from a completed loop result. */
1940
- declare function buildLoopProvenanceRecord<TArtifact, TScenario extends Scenario>(args: BuildLoopProvenanceArgs<TArtifact, TScenario>): LoopProvenanceRecord;
1941
- /** Digest the exact campaign fields that can affect a measured comparison. */
1942
- declare function campaignMeasurementDigest<TArtifact, TScenario extends Scenario>(campaign: CampaignResult<TArtifact, TScenario>): `sha256:${string}`;
1943
- /** Recompute and validate the self-addressed durable record. */
1944
- declare function verifyLoopProvenanceRecord(record: LoopProvenanceRecord): LoopProvenanceRecord;
1945
- /** SHA-256 over the RFC 8785 canonical JSON of `value`. Throws
1946
- * `LedgerCanonicalizationError` for a value with no canonical form. */
1947
- declare function canonicalDigest(value: unknown): `sha256:${string}`;
1948
- /**
1949
- * Build the loop's OTLP-ingestable spans from a provenance record. One root
1950
- * span per loop (`tangle.runId`), one span per generation, one span per
1951
- * candidate (carrying its surfaceHash + label), and one span for the gate
1952
- * decision (carrying reasons + delta + lift). Candidate + gate spans pivot on
1953
- * the same `tangle.runId` / `tangle.generation` attributes `/adapters/otel`
1954
- * reads, so the hosted collector reconstructs the full tree.
1955
- *
1956
- * Times are synthesized monotonically off a single base so the span tree is
1957
- * orderable; the substrate does not retain per-candidate wall-clock starts.
1958
- */
1959
- declare function loopProvenanceSpans(record: LoopProvenanceRecord, opts?: {
1960
- baseTimeMs?: number;
1961
- }): TraceSpanEvent[];
1962
- /** Canonical durable paths under the run dir. */
1963
- declare function provenanceRecordPath(runDir: string): string;
1964
- /**
1965
- * Canonical path for the durable OTLP spans JSONL file under a loop run directory.
1966
- */
1967
- declare function provenanceSpansPath(runDir: string): string;
1968
- interface EmitLoopProvenanceResult {
1969
- record: LoopProvenanceRecord;
1970
- spans: TraceSpanEvent[];
1971
- /** Absolute paths the record + spans were written to, when storage persists. */
1972
- recordPath: string;
1973
- spansPath: string;
1974
- }
1975
- interface EmitLoopProvenanceArgs<TArtifact, TScenario extends Scenario> extends BuildLoopProvenanceArgs<TArtifact, TScenario> {
1976
- /** Storage the record + spans are written through. */
1977
- storage: CampaignStorage;
1978
- /** When set, the spans are also shipped to the hosted `/v1/ingest/traces`
1979
- * endpoint so the collector receives the full loop, not just `cost.*`. */
1980
- hostedClient?: HostedClient;
1981
- }
1982
- /**
1983
- * Build the provenance record + OTel spans and persist them durably under the
1984
- * run dir (and ship spans to a hosted collector when one is wired). Returns
1985
- * both artifacts so the caller can assert on / re-derive from them.
1986
- *
1987
- * Fail-loud: the durable write throws on storage failure (a swallowed write is
1988
- * exactly the "emitted but lost" failure this closes). The hosted span ship is
1989
- * the one best-effort leg — its failure is logged, not thrown, so an offline
1990
- * collector never fails the loop (the durable artifact is the source of truth).
1991
- */
1992
- declare function emitLoopProvenance<TArtifact, TScenario extends Scenario>(args: EmitLoopProvenanceArgs<TArtifact, TScenario>): Promise<EmitLoopProvenanceResult>;
1993
- //#endregion
1994
- export { readExternalOptimizerObservationArtifact as $, SearchLedger as $t, CanaryOptions as A, RunEvalOptions as An, SearchHistoryCoverageRow as At, OptimizationMethodResult as B, CacheIssueReason as Bn, OpenSearchLedgerOptions as Bt, RedTeamReport as C, CrowdedFrontierParentOptions as Cn, GepaCandidatePopulationCandidate as Ct, CanaryAlert as D, OpenAutoPrOptions as Dn, CreateSearchHistoryReceiptInput as Dt, scoreRedTeamOutput as E, crowdedFrontierParent as En, readGepaCandidatePopulationArtifact as Et, OptimizationMethod as F, runCampaign as Fn, assertSearchHistoryMatchesReplay as Ft, combineComparisonCosts as G, cellCachePath as Gn, SearchCandidateLineage as Gt, OptimizationMethodScore as H, readCachedCell as Hn, SearchArtifactRef as Ht, OptimizationMethodComparison as I, CampaignRunPlan as In, createSearchHistoryReceipt as It, optimizationTokenUsageFromSummary as J, fsCampaignStorage as Jn, SearchCandidateSlotClosedEvent as Jt, compareOptimizationMethods as K, CampaignStorage as Kn, SearchCandidateRegisteredEvent as Kt, OptimizationMethodInput as L, CampaignRunPlanCell as Ln, searchHistoryCoverageRow as Lt, runCanaries as M, CampaignCellFailureReceipt as Mn, SearchHistoryReceipt as Mt, CompareOptimizationMethodsOptions as N, CampaignCellRetryPolicy as Nn, SearchHistoryRequiredError as Nt, CanaryEvaluation as O, OpenAutoPrResult as On, SearchHistoryAuditSummary as Ot, ComparisonCost as P, RunCampaignOptions as Pn, assertCompleteSearchHistory as Pt, ExternalOptimizerSubmittedCandidate as Q, SearchFailureReason as Qt, OptimizationMethodPairwise as R, PlanCampaignRunOptions as Rn, verifySearchHistoryReceipt as Rt, RedTeamFinding as S, validateSearchLedgerEvent as Sn, GepaCandidatePopulationArtifact as St, redTeamReport as T, ParentSelector as Tn, GepaCandidateSelectionScore as Tt, OptimizationPackageSource as U, CellScheduleSlot as Un, SearchAttemptAccounting as Ut, OptimizationMethodRunOptions as V, CacheRead as Vn, SearchAccountingAudit as Vt, OptimizationTokenUsage as W, buildCellSchedule as Wn, SearchCandidateDecidedEvent as Wt, ExternalOptimizerObservationArtifact as X, SearchCompletedEvent as Xt, ExternalOptimizerExecutionSummary as Y, inMemoryCampaignStorage as Yn, SearchCandidateSurface as Yt, ExternalOptimizerObservationSummary as Z, SearchCostAccounting as Zt, provenanceSpansPath as _, SearchSurfaceKind as _n, SearchLedgerBinding as _t, LoopProvenanceBackend as a, SearchLedgerTrustedHeadMode as an, quotaExhaustedUntil as at, RedTeamCase as b, SearchTokenAccounting as bn, SearchRunIdentity as bt, LoopProvenanceOptimizationMethod as c, SearchOperationRecordedEvent as cn, RunImprovementLoopResult as ct, campaignMeasurementDigest as d, SearchPlannedEvent as dn, RunOptimizationOptions as dt, SearchLedgerAppendResult as en, LlmJudgeDimension as et, canonicalDigest as f, SearchPlannedOperation as fn, RunOptimizationResult as ft, provenanceRecordPath as g, SearchSurfaceEvidence as gn, SearchExecutionIdentity as gt, loopProvenanceSpans as h, SearchSurfaceEffect as hn, ProposedSearchCandidate as ht, LoopProvenanceArgsFromResult as i, SearchLedgerReplay as in, isTransientTransportFailure as it, CanaryReport as j, runEval as jn, SearchHistoryPolicy as jt, CanaryKind as k, openAutoPr as kn, SearchHistoryCoverage as kt, LoopProvenanceRecord as l, SearchPlan as ln, runImprovementLoop as lt, loopProvenanceArgsFromResult as m, SearchSourceRef as mn, MeasuredSearchCandidate as mt, EmitLoopProvenanceArgs as n, SearchLedgerEvent as nn, llmJudge as nt, LoopProvenanceCandidate as o, SearchModelIdentity as on, transientDispatchFailure as ot, emitLoopProvenance as p, SearchPlannedTask as pn, runOptimization as pt, costFromLedgerSummary as q, createRunCostLedger as qn, SearchCandidateSlot as qt, EmitLoopProvenanceResult as r, SearchLedgerHash as rn, TransientFailureOptions as rt, LoopProvenanceEvidence as s, SearchOperationKind as sn, RunImprovementLoopOptions as st, BuildLoopProvenanceArgs as t, SearchLedgerEntry as tn, LlmJudgeOptions as tt, buildLoopProvenanceRecord as u, SearchPlanExtendedEvent as un, PremeasuredOptimizationBaseline as ut, verifyLoopProvenanceRecord as v, SearchTaskAttemptedEvent as vn, SearchRecorder as vt, redTeamDataset as w, ParentSelectionContext as wn, GepaCandidatePopulationSummary as wt, RedTeamCategory as x, openSearchLedger as xn, recordCandidatePopulationSearch as xt, DEFAULT_RED_TEAM_CORPUS as y, SearchTaskOutcome as yn, SearchRecorderOptions as yt, OptimizationMethodProvenance as z, planCampaignRun as zn, FileSearchLedger as zt };
1995
- //# sourceMappingURL=provenance-CRY67X50.d.ts.map