@tangle-network/agent-eval 0.173.3 → 0.174.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +31 -0
- package/dist/{proposal-findings-bko3GGy-.js → abort-signal-CtzAM_sJ.js} +11 -11
- package/dist/abort-signal-CtzAM_sJ.js.map +1 -0
- package/dist/adapters/http.d.ts +2 -2
- package/dist/agent-profile-_xPxqVJt.d.ts +488 -0
- package/dist/agent-profile-_xPxqVJt.d.ts.map +1 -0
- package/dist/analyst/index.d.ts +7 -9
- package/dist/analyst/index.d.ts.map +1 -1
- package/dist/analyst/index.js +8 -8
- package/dist/{benchmark-C4wk_Sjr.js → benchmark-DQKzykkO.js} +2 -2
- package/dist/{benchmark-C4wk_Sjr.js.map → benchmark-DQKzykkO.js.map} +1 -1
- package/dist/{benchmark-command-BY9oscke.js → benchmark-command-mZIlR-ra.js} +13 -13
- package/dist/{benchmark-command-BY9oscke.js.map → benchmark-command-mZIlR-ra.js.map} +1 -1
- package/dist/benchmarks/index.d.ts +3 -4
- package/dist/benchmarks/index.d.ts.map +1 -1
- package/dist/benchmarks/index.js +3 -3
- package/dist/campaign/index.d.ts +5 -9
- package/dist/campaign/index.js +7 -7
- package/dist/{campaign-B3kPMU8S.js → campaign-BzMSCejE.js} +8 -8
- package/dist/{campaign-B3kPMU8S.js.map → campaign-BzMSCejE.js.map} +1 -1
- package/dist/cli.js +1 -1
- package/dist/{client-DlqdbM7n.d.ts → client-vyYQg3bm.d.ts} +2 -2
- package/dist/{client-DlqdbM7n.d.ts.map → client-vyYQg3bm.d.ts.map} +1 -1
- package/dist/contract/index.d.ts +9 -10
- package/dist/contract/index.js +8 -8
- package/dist/{default-registry-B0bKikCb.js → default-registry-CrAp0pYq.js} +4 -4
- package/dist/{default-registry-B0bKikCb.js.map → default-registry-CrAp0pYq.js.map} +1 -1
- package/dist/{default-registry-BKwc8bN5.d.ts → default-registry-FfNzaUHV.d.ts} +3 -3
- package/dist/{default-registry-BKwc8bN5.d.ts.map → default-registry-FfNzaUHV.d.ts.map} +1 -1
- package/dist/{define-agent-eval-CY6qdlGV.d.ts → define-agent-eval-V1jQyCDR.d.ts} +102 -11
- package/dist/define-agent-eval-V1jQyCDR.d.ts.map +1 -0
- package/dist/{define-agent-eval-8h3lXXee.js → define-agent-eval-ox5McL6e.js} +331 -144
- package/dist/define-agent-eval-ox5McL6e.js.map +1 -0
- package/dist/{dspy-rlm-engine-CF0t2ITD.js → dspy-rlm-engine-Caz2pl4L.js} +3 -3
- package/dist/{dspy-rlm-engine-CF0t2ITD.js.map → dspy-rlm-engine-Caz2pl4L.js.map} +1 -1
- package/dist/{engine-DhFir3Ys.d.ts → engine-CvW_I72-.d.ts} +2 -2
- package/dist/{engine-DhFir3Ys.d.ts.map → engine-CvW_I72-.d.ts.map} +1 -1
- package/dist/experiment/index.d.ts +1 -4
- package/dist/experiment/index.d.ts.map +1 -1
- package/dist/{external-optimizer-process-BwITA9Jp.js → external-optimizer-process-CxnFL1hd.js} +2 -2
- package/dist/{external-optimizer-process-BwITA9Jp.js.map → external-optimizer-process-CxnFL1hd.js.map} +1 -1
- package/dist/{external-optimizer-subprocess-wBWeoG6A.js → external-optimizer-subprocess-CQi27uEI.js} +2 -2
- package/dist/{external-optimizer-subprocess-wBWeoG6A.js.map → external-optimizer-subprocess-CQi27uEI.js.map} +1 -1
- package/dist/fuzz.js +1 -1
- package/dist/fuzz.js.map +1 -1
- package/dist/hosted/index.d.ts +1 -1
- package/dist/{index-BQqOjerE.d.ts → index-BTrx5s8m.d.ts} +8 -9
- package/dist/index-BTrx5s8m.d.ts.map +1 -0
- package/dist/{index-D0Db5X-4.d.ts → index-Bn-nlnSV.d.ts} +4 -4
- package/dist/{index-D0Db5X-4.d.ts.map → index-Bn-nlnSV.d.ts.map} +1 -1
- package/dist/index-DKXuBPXf.d.ts +3840 -0
- package/dist/index-DKXuBPXf.d.ts.map +1 -0
- package/dist/index.d.ts +11 -13
- package/dist/index.d.ts.map +1 -1
- package/dist/index.js +10 -10
- package/dist/{kind-factory-gP6lDySe.js → kind-factory-BLvL-E44.js} +2 -2
- package/dist/{kind-factory-gP6lDySe.js.map → kind-factory-BLvL-E44.js.map} +1 -1
- package/dist/{llm-judge-BfqMFo4h.js → llm-judge-DmNaBrXB.js} +2541 -2435
- package/dist/llm-judge-DmNaBrXB.js.map +1 -0
- package/dist/{matrix-DGu8KhSs.d.ts → matrix-CJtXz1ky.d.ts} +2 -2
- package/dist/{matrix-DGu8KhSs.d.ts.map → matrix-CJtXz1ky.d.ts.map} +1 -1
- package/dist/multishot/golden/index.d.ts +1 -1
- package/dist/multishot/index.d.ts +2 -2
- package/dist/openapi.json +1 -1
- package/dist/{produced-state-D91uDvQw.js → produced-state-B8mw6zj9.js} +2 -2
- package/dist/{produced-state-D91uDvQw.js.map → produced-state-B8mw6zj9.js.map} +1 -1
- package/dist/rl.d.ts +1 -1
- package/dist/rl.d.ts.map +1 -1
- package/dist/rl.js.map +1 -1
- package/dist/{semantic-concept-judge-Dok7_35a.js → semantic-concept-judge-E3s_fEjB.js} +3 -3
- package/dist/{semantic-concept-judge-Dok7_35a.js.map → semantic-concept-judge-E3s_fEjB.js.map} +1 -1
- package/dist/{skillopt-optimization-method-DDw3v3gA.js → skillopt-optimization-method-f7399oGb.js} +5 -5
- package/dist/{skillopt-optimization-method-DDw3v3gA.js.map → skillopt-optimization-method-f7399oGb.js.map} +1 -1
- package/dist/statistical-heldout-Cqb73yE9.d.ts +1127 -0
- package/dist/statistical-heldout-Cqb73yE9.d.ts.map +1 -0
- package/dist/{store-otlp-Dow0pk_5.js → store-otlp-DV_H2HDu.js} +2 -2
- package/dist/{store-otlp-Dow0pk_5.js.map → store-otlp-DV_H2HDu.js.map} +1 -1
- package/dist/{store-tool-spans-CCZNsihA.d.ts → store-tool-spans-4o55ABER.d.ts} +3 -3
- package/dist/{store-tool-spans-CCZNsihA.d.ts.map → store-tool-spans-4o55ABER.d.ts.map} +1 -1
- package/dist/{store-tool-spans-CeNj_m2L.js → store-tool-spans-B9tjys_h.js} +3 -3
- package/dist/{store-tool-spans-CeNj_m2L.js.map → store-tool-spans-B9tjys_h.js.map} +1 -1
- package/dist/{task-failure-attributes-CZjZeBsY.js → task-failure-attributes-CUy9mkIY.js} +2 -2
- package/dist/{task-failure-attributes-CZjZeBsY.js.map → task-failure-attributes-CUy9mkIY.js.map} +1 -1
- package/dist/{tool-groups-Cp4Xdzrp.d.ts → tool-groups-DAe1t6zb.d.ts} +2 -2
- package/dist/tool-groups-DAe1t6zb.d.ts.map +1 -0
- package/dist/trace-repair/index.d.ts +1 -1
- package/dist/traces.d.ts +2 -2
- package/dist/traces.js +4 -4
- package/dist/{types-Ba5UQyVD.d.ts → types-BJz2CPTM.d.ts} +2 -2
- package/dist/{types-Ba5UQyVD.d.ts.map → types-BJz2CPTM.d.ts.map} +1 -1
- package/dist/{types-CiWITkGo.js → types-DQ0e2E7y.js} +2 -2
- package/dist/types-DQ0e2E7y.js.map +1 -0
- package/dist/{types-BDV4PiMR.d.ts → types-Dd1ejaeI.d.ts} +2 -2
- package/dist/{types-BDV4PiMR.d.ts.map → types-Dd1ejaeI.d.ts.map} +1 -1
- package/docs/campaign-proposers.md +42 -0
- package/package.json +1 -1
- package/dist/agent-profile-B9_GGsG8.d.ts +0 -84
- package/dist/agent-profile-B9_GGsG8.d.ts.map +0 -1
- package/dist/backend-integrity-CeuTgqsd.d.ts +0 -280
- package/dist/backend-integrity-CeuTgqsd.d.ts.map +0 -1
- package/dist/benchmark-BjLGkfnN.d.ts +0 -236
- package/dist/benchmark-BjLGkfnN.d.ts.map +0 -1
- package/dist/define-agent-eval-8h3lXXee.js.map +0 -1
- package/dist/define-agent-eval-CY6qdlGV.d.ts.map +0 -1
- package/dist/external-optimizer-contracts-CQCpyrIL.d.ts +0 -172
- package/dist/external-optimizer-contracts-CQCpyrIL.d.ts.map +0 -1
- package/dist/heldout-gate-Df5hsqmm.d.ts +0 -453
- package/dist/heldout-gate-Df5hsqmm.d.ts.map +0 -1
- package/dist/index-BQqOjerE.d.ts.map +0 -1
- package/dist/index-CFDffsKz.d.ts +0 -1135
- package/dist/index-CFDffsKz.d.ts.map +0 -1
- package/dist/llm-judge-BfqMFo4h.js.map +0 -1
- package/dist/power-preflight-Ptse_Kq7.d.ts +0 -117
- package/dist/power-preflight-Ptse_Kq7.d.ts.map +0 -1
- package/dist/pre-registration-BoI4ucR3.d.ts +0 -592
- package/dist/pre-registration-BoI4ucR3.d.ts.map +0 -1
- package/dist/promotion-policy-CvMda3kU.d.ts +0 -134
- package/dist/promotion-policy-CvMda3kU.d.ts.map +0 -1
- package/dist/proposal-findings-bko3GGy-.js.map +0 -1
- package/dist/provenance-CRY67X50.d.ts +0 -1995
- package/dist/provenance-CRY67X50.d.ts.map +0 -1
- package/dist/statistical-heldout-DTyB_6-1.d.ts +0 -295
- package/dist/statistical-heldout-DTyB_6-1.d.ts.map +0 -1
- package/dist/tool-groups-Cp4Xdzrp.d.ts.map +0 -1
- package/dist/types-CiWITkGo.js.map +0 -1
|
@@ -1,1995 +0,0 @@
|
|
|
1
|
-
import { c as CostLedgerHandle, f as CostLedgerSummary, m as CostReceipt, o as CostLedger, p as CostProvenance } from "./cost-ledger-DbQdN3nO.js";
|
|
2
|
-
import { a as RunRecord } from "./run-record-DTv1MdjK.js";
|
|
3
|
-
import { _ as ProposalFinding } from "./types-DN2WdT5S.js";
|
|
4
|
-
import { p as ChatClient } from "./types-gvRsyJLh.js";
|
|
5
|
-
import { B as ScoredSurfaceOutcome, C as JudgeDimension, H as SurfaceProposer, N as ParetoParent, R as Scenario, S as JudgeConfig, _ as GateDecision, a as CampaignResult, b as GenerationRecord, c as CampaignTraceWriter, d as DispatchContext, f as DispatchFn, g as GateContribution, j as MutableSurface, k as LabeledScenarioStore, p as Gate, r as CampaignCellResult, v as GateResult, y as GenerationCandidate } from "./types-Ba5UQyVD.js";
|
|
6
|
-
import { m as LedgerTrustedHeadRemoval, p as LedgerTrustedHead } from "./index-D-UdhAmg.js";
|
|
7
|
-
import { r as LedgerHash } from "./canonical-CFpojCN5.js";
|
|
8
|
-
import { _ as ExternalTextCandidate, g as ExternalOptimizerWireCounts, o as ExternalOptimizerEvaluationObservation } from "./external-optimizer-contracts-CQCpyrIL.js";
|
|
9
|
-
import { r as DatasetScenario, t as Dataset } from "./dataset-DQqhOCPt.js";
|
|
10
|
-
import { g as TraceSpanEvent, t as HostedClient } from "./client-DlqdbM7n.js";
|
|
11
|
-
import { z } from "zod";
|
|
12
|
-
//#region src/campaign/storage.d.ts
|
|
13
|
-
/**
|
|
14
|
-
* `CampaignStorage` — the filesystem seam `runCampaign` writes through
|
|
15
|
-
* (run/cell dirs, the resumability cache, per-cell artifacts, trace spans).
|
|
16
|
-
*
|
|
17
|
-
* The default (`fsCampaignStorage`) is the Node filesystem — identical
|
|
18
|
-
* behavior to the inline `node:fs` calls it replaces, so existing CLI
|
|
19
|
-
* consumers are unaffected. `inMemoryCampaignStorage` keeps everything in a
|
|
20
|
-
* `Map`, so the substrate runs in environments WITHOUT a filesystem
|
|
21
|
-
* (Cloudflare Workers, Deno Deploy, other edge runtimes) — the campaign
|
|
22
|
-
* still produces its `CampaignResult` (cells + aggregates) in memory;
|
|
23
|
-
* artifacts/traces simply aren't persisted to disk.
|
|
24
|
-
*
|
|
25
|
-
* Paths are opaque keys to the in-memory adapter — it does not parse them,
|
|
26
|
-
* so the same `join(...)`-built paths work unchanged across both adapters.
|
|
27
|
-
*/
|
|
28
|
-
interface CampaignStorage {
|
|
29
|
-
/** Ensure a directory exists (recursive). No-op for in-memory. */
|
|
30
|
-
ensureDir(dir: string): void;
|
|
31
|
-
/** Does this path exist (as a written file or an ensured dir)? */
|
|
32
|
-
exists(path: string): boolean;
|
|
33
|
-
/** Read a UTF-8 file; `undefined` when missing or unreadable. */
|
|
34
|
-
read(path: string): string | undefined;
|
|
35
|
-
/** Write a file (string or bytes). Parent dir is assumed ensured. */
|
|
36
|
-
write(path: string, content: string | Uint8Array): void;
|
|
37
|
-
/** Append only when the current UTF-8 byte length matches `expectedBytes`.
|
|
38
|
-
* Returns the new length, or undefined when another writer won. */
|
|
39
|
-
append(path: string, content: string, expectedBytes: number): number | undefined;
|
|
40
|
-
}
|
|
41
|
-
/** Node-filesystem storage — the default. Lazily requires `node:fs` so the
|
|
42
|
-
* module imports cleanly in non-Node runtimes (where the caller passes
|
|
43
|
-
* `inMemoryCampaignStorage` instead and never constructs this).
|
|
44
|
-
*
|
|
45
|
-
* `createRequire(import.meta.url)` is the ESM-native lazy require — a bare
|
|
46
|
-
* `require` is a ReferenceError under `"type": "module"`, which is exactly
|
|
47
|
-
* the shape this package publishes. */
|
|
48
|
-
declare function fsCampaignStorage(): CampaignStorage;
|
|
49
|
-
/** In-memory storage for filesystem-less runtimes. Artifacts + trace spans
|
|
50
|
-
* live in a `Map` for the duration of the run; the `CampaignResult` is
|
|
51
|
-
* fully populated, but nothing is persisted to disk. */
|
|
52
|
-
declare function inMemoryCampaignStorage(): CampaignStorage;
|
|
53
|
-
/** Open the durable spend account stored beside a logical run. */
|
|
54
|
-
declare function createRunCostLedger(input: {
|
|
55
|
-
storage: CampaignStorage;
|
|
56
|
-
runDir: string;
|
|
57
|
-
costCeilingUsd?: number;
|
|
58
|
-
/** Set false for read-only inspection that must not create the run directory. */
|
|
59
|
-
ensureRunDir?: boolean;
|
|
60
|
-
}): CostLedger;
|
|
61
|
-
//#endregion
|
|
62
|
-
//#region src/campaign/cell-schedule.d.ts
|
|
63
|
-
declare function buildCellSchedule<TScenario extends Scenario>(scenarios: TScenario[], seed: number, reps: number): Array<{
|
|
64
|
-
scenario: TScenario;
|
|
65
|
-
rep: number;
|
|
66
|
-
cellId: string;
|
|
67
|
-
cellSeed: number;
|
|
68
|
-
}>;
|
|
69
|
-
type CellScheduleSlot<TScenario extends Scenario> = ReturnType<typeof buildCellSchedule<TScenario>>[number];
|
|
70
|
-
declare function cellCachePath(runDir: string, cellId: string): string;
|
|
71
|
-
//#endregion
|
|
72
|
-
//#region src/campaign/cell-cache.d.ts
|
|
73
|
-
type CacheIssueReason = 'missing' | 'manifest-mismatch' | 'cell-mismatch' | 'missing-cost-provenance' | 'invalid-cost-provenance' | 'invalid-cost-receipts' | 'corrupt';
|
|
74
|
-
type CacheRead<TArtifact> = {
|
|
75
|
-
status: 'hit';
|
|
76
|
-
cell: CampaignCellResult<TArtifact>;
|
|
77
|
-
} | {
|
|
78
|
-
status: 'miss';
|
|
79
|
-
reason: CacheIssueReason;
|
|
80
|
-
};
|
|
81
|
-
declare function readCachedCell<TArtifact>(args: {
|
|
82
|
-
storage: CampaignStorage;
|
|
83
|
-
cachePath: string;
|
|
84
|
-
cellId: string;
|
|
85
|
-
manifestHash: string;
|
|
86
|
-
}): CacheRead<TArtifact>;
|
|
87
|
-
//#endregion
|
|
88
|
-
//#region src/campaign/plan-campaign-run.d.ts
|
|
89
|
-
interface CampaignRunPlanCell {
|
|
90
|
-
cellId: string;
|
|
91
|
-
scenarioId: string;
|
|
92
|
-
rep: number;
|
|
93
|
-
seed: number;
|
|
94
|
-
cachePath: string;
|
|
95
|
-
status: 'cached' | 'run' | 'blocked';
|
|
96
|
-
reason?: CacheIssueReason | 'resumable-off';
|
|
97
|
-
}
|
|
98
|
-
interface CampaignRunPlan {
|
|
99
|
-
manifestHash: string;
|
|
100
|
-
splitDigest: `sha256:${string}`;
|
|
101
|
-
totalCells: number;
|
|
102
|
-
cellsCached: number;
|
|
103
|
-
cellsBlocked: number;
|
|
104
|
-
cellsToRun: number;
|
|
105
|
-
cells: CampaignRunPlanCell[];
|
|
106
|
-
}
|
|
107
|
-
interface PlanCampaignRunOptions<TScenario extends Scenario, TArtifact> {
|
|
108
|
-
scenarios: TScenario[];
|
|
109
|
-
dispatch?: DispatchFn<TScenario, TArtifact>;
|
|
110
|
-
dispatchRef?: string;
|
|
111
|
-
judges?: JudgeConfig<TArtifact, TScenario>[];
|
|
112
|
-
seed?: number;
|
|
113
|
-
reps?: number;
|
|
114
|
-
resumable?: boolean;
|
|
115
|
-
/** See RunCampaignOptions.rerunInvalidCachedCells. */
|
|
116
|
-
rerunInvalidCachedCells?: boolean;
|
|
117
|
-
runDir: string;
|
|
118
|
-
/** Subject repo for the shared run-dir root (see RunCampaignOptions.repo). */
|
|
119
|
-
repo?: string;
|
|
120
|
-
storage?: CampaignStorage;
|
|
121
|
-
/** Spend account used to validate cached receipt identities. */
|
|
122
|
-
costLedger?: CostLedgerHandle;
|
|
123
|
-
/** Receipt tags used by the campaign that produced the cached cells. */
|
|
124
|
-
costTags?: Readonly<Record<string, string>>;
|
|
125
|
-
}
|
|
126
|
-
/**
|
|
127
|
-
* Plan a campaign WITHOUT dispatching: computes the manifest hash and the per-cell
|
|
128
|
-
* run-vs-cached schedule so callers can preview cost and resumability before spending.
|
|
129
|
-
*/
|
|
130
|
-
declare function planCampaignRun<TScenario extends Scenario, TArtifact>(opts: PlanCampaignRunOptions<TScenario, TArtifact>): CampaignRunPlan;
|
|
131
|
-
//#endregion
|
|
132
|
-
//#region src/campaign/run-campaign.d.ts
|
|
133
|
-
interface RunCampaignOptions<TScenario extends Scenario, TArtifact> {
|
|
134
|
-
scenarios: TScenario[];
|
|
135
|
-
dispatch: DispatchFn<TScenario, TArtifact>;
|
|
136
|
-
/** Abort active dispatches when the owning operation is cancelled. */
|
|
137
|
-
signal?: AbortSignal;
|
|
138
|
-
/**
|
|
139
|
-
* Stable identity for the dispatch behavior, included in the manifest/cache
|
|
140
|
-
* key. Set this when the same function name can run different models,
|
|
141
|
-
* prompts, tools, or external config.
|
|
142
|
-
*/
|
|
143
|
-
dispatchRef?: string;
|
|
144
|
-
judges?: JudgeConfig<TArtifact, TScenario>[];
|
|
145
|
-
/** Required for reproducibility. Default 42. */
|
|
146
|
-
seed?: number;
|
|
147
|
-
/** Per-scenario replicates for CI bands. Default 1; raise to 5+ for
|
|
148
|
-
* bootstrap-tight intervals on critical eval. */
|
|
149
|
-
reps?: number;
|
|
150
|
-
/** When true (default), completed cells are cached by
|
|
151
|
-
* (manifestHash, scenarioId, rep, generation). Re-runs skip cached cells. */
|
|
152
|
-
resumable?: boolean;
|
|
153
|
-
/**
|
|
154
|
-
* Optional explicit cell selection. The campaign manifest and split digest
|
|
155
|
-
* still describe the complete declared scenario × replicate design; this
|
|
156
|
-
* only limits the rows executed by this invocation.
|
|
157
|
-
*/
|
|
158
|
-
cellFilter?: (input: {
|
|
159
|
-
scenario: TScenario;
|
|
160
|
-
rep: number;
|
|
161
|
-
}) => boolean;
|
|
162
|
-
/**
|
|
163
|
-
* Reuse a cached cell that has an error instead of dispatching it again.
|
|
164
|
-
* The default retries failed cells, preserving normal campaign behaviour.
|
|
165
|
-
*/
|
|
166
|
-
reuseFailedCells?: boolean;
|
|
167
|
-
/**
|
|
168
|
-
* Explicitly rerun only cached cells whose saved result is unreadable or
|
|
169
|
-
* has missing/invalid cost provenance. Valid cached cells remain reusable.
|
|
170
|
-
* Default false refuses to begin work when any such cache entry exists.
|
|
171
|
-
*/
|
|
172
|
-
rerunInvalidCachedCells?: boolean;
|
|
173
|
-
/** Optional store — when present, every artifact + judge score is captured
|
|
174
|
-
* with the configured `captureSource`. Capture is default ON; pass `'off'`
|
|
175
|
-
* to disable. */
|
|
176
|
-
labeledStore?: LabeledScenarioStore | 'off';
|
|
177
|
-
captureSource?: 'production-trace' | 'eval-run' | 'manual' | 'red-team' | 'synthetic';
|
|
178
|
-
captureSourceVersionHash?: string;
|
|
179
|
-
/** Hard spend cap. Each paid call reserves its enforced maximum before dispatch. */
|
|
180
|
-
costCeiling?: number;
|
|
181
|
-
/** Shared spend account. Improvement loops pass one ledger through every
|
|
182
|
-
* campaign so the ceiling and returned total are run-wide. */
|
|
183
|
-
costLedger?: CostLedgerHandle;
|
|
184
|
-
/** Attribution label for receipts recorded by this campaign. */
|
|
185
|
-
costPhase?: string;
|
|
186
|
-
/** Additional immutable receipt tags supplied by an owning workflow. */
|
|
187
|
-
costTags?: Readonly<Record<string, string>>;
|
|
188
|
-
/** Max concurrent cells. Default 2. */
|
|
189
|
-
maxConcurrency?: number;
|
|
190
|
-
/**
|
|
191
|
-
* Stop after the first dispatch or judge error. The failed cell is persisted
|
|
192
|
-
* before active sibling cells are aborted and drained, then the campaign
|
|
193
|
-
* rejects with the exact error thrown by that dispatch or judge.
|
|
194
|
-
* Default false preserves the normal behavior of returning failed cells and
|
|
195
|
-
* continuing the remaining schedule.
|
|
196
|
-
* With `cellRetry`, a retryable failure is not an error yet: this abort
|
|
197
|
-
* fires only when a cell's final attempt fails.
|
|
198
|
-
*/
|
|
199
|
-
abortOnCellError?: boolean;
|
|
200
|
-
/**
|
|
201
|
-
* Opt-in bounded in-run retry of a failed cell. Absent by default: a failed
|
|
202
|
-
* cell is final on its first attempt. A retried attempt re-runs the SAME
|
|
203
|
-
* slot — same `cellId`, same `seed`, same cost tags — so the schedule,
|
|
204
|
-
* manifest, and pairing are unchanged. An attempt that failed because the
|
|
205
|
-
* campaign was cancelled is never retried.
|
|
206
|
-
*/
|
|
207
|
-
cellRetry?: CampaignCellRetryPolicy;
|
|
208
|
-
/**
|
|
209
|
-
* Per-cell dispatch deadline in ms. A `dispatch` that neither resolves nor
|
|
210
|
-
* rejects within this window is a hang (a stalled model request, an
|
|
211
|
-
* exhausted runtime resource, a backend that never closes its stream). When
|
|
212
|
-
* set, the cell's `ctx.signal` is aborted. A dispatch that stops is recorded
|
|
213
|
-
* as an error (`dispatch exceeded <N>ms`). A dispatch that ignores
|
|
214
|
-
* cancellation rejects the campaign without publishing incomplete cost data.
|
|
215
|
-
* `undefined`/`0` means unbounded.
|
|
216
|
-
*/
|
|
217
|
-
dispatchTimeoutMs?: number;
|
|
218
|
-
/**
|
|
219
|
-
* Time allowed for an aborted dispatch and its paid calls to stop before the
|
|
220
|
-
* campaign rejects without producing a result. Default 5 seconds.
|
|
221
|
-
*/
|
|
222
|
-
dispatchShutdownTimeoutMs?: number;
|
|
223
|
-
/** Required: where artifacts + traces land. A bare name (not an absolute path)
|
|
224
|
-
* resolves to the shared `~/.tangle/traces/<repo>/runs/<name>` root so run
|
|
225
|
-
* bundles never pollute a repo working tree. Pass an absolute path to override. */
|
|
226
|
-
runDir: string;
|
|
227
|
-
/** Subject repo for the shared run-dir root (defaults to the CWD basename).
|
|
228
|
-
* Only consulted when `runDir` is a bare name. */
|
|
229
|
-
repo?: string;
|
|
230
|
-
/** Tracing posture. Default is the substrate's `FileSystemTraceStore` rooted
|
|
231
|
-
* at `<runDir>/traces/`. `'off'` disables capture entirely — substrate
|
|
232
|
-
* refuses this when the caller wires `autoOnPromote !== 'none'`. */
|
|
233
|
-
tracing?: 'on' | 'off';
|
|
234
|
-
/**
|
|
235
|
-
* Per-cell usage expectation — the early, fine-grained sibling of the
|
|
236
|
-
* batch `assertRealBackend` guard. A cell that produced an artifact (no
|
|
237
|
-
* error) but reported `costUsd === 0` AND zero tokens is a stub: the
|
|
238
|
-
* dispatch never reported LLM activity via `ctx.cost`. Modes:
|
|
239
|
-
* - `'warn'` (default) — log the offending cell loudly, keep going.
|
|
240
|
-
* - `'assert'` — throw `BackendIntegrityError` on the first such cell
|
|
241
|
-
* (fail-fast; recommended for CI campaigns expecting real LLM calls).
|
|
242
|
-
* - `'off'` — no check (replay / deterministic-only / offline analysis).
|
|
243
|
-
*/
|
|
244
|
-
expectUsage?: 'assert' | 'warn' | 'off';
|
|
245
|
-
/** Test seam — override the wall clock for deterministic tests. */
|
|
246
|
-
now?: () => Date;
|
|
247
|
-
/** Test seam — override per-cell trace writer factory. */
|
|
248
|
-
buildTraceWriter?: (cellId: string, dir: string) => CampaignTraceWriter;
|
|
249
|
-
/** Storage backend for run/cell dirs, the resumability cache, artifacts,
|
|
250
|
-
* and trace spans. Default: the Node filesystem (`fsCampaignStorage`).
|
|
251
|
-
* Pass `inMemoryCampaignStorage()` to run in a filesystem-less runtime
|
|
252
|
-
* (Cloudflare Workers, Deno, edge) — the `CampaignResult` is still
|
|
253
|
-
* produced; artifacts/traces just aren't persisted to disk. */
|
|
254
|
-
storage?: CampaignStorage;
|
|
255
|
-
/**
|
|
256
|
-
* Optional per-cell placement strategy. Returns an opaque string the
|
|
257
|
-
* substrate forwards as `ctx.placement` to the Dispatch — placement-aware
|
|
258
|
-
* Dispatches (e.g. `httpDispatch` from `/adapters/http`) use it to route
|
|
259
|
-
* each cell to the right worker, region, or sandbox. When unset, every
|
|
260
|
-
* cell receives `ctx.placement = undefined` and behaves identically to
|
|
261
|
-
* the in-process case.
|
|
262
|
-
*
|
|
263
|
-
* @example
|
|
264
|
-
* cellPlacement: ({ scenario }) => scenario.tags?.includes('eu') ? 'eu-west' : 'us-east'
|
|
265
|
-
*/
|
|
266
|
-
cellPlacement?: (input: {
|
|
267
|
-
scenario: TScenario;
|
|
268
|
-
rep: number;
|
|
269
|
-
generation?: number;
|
|
270
|
-
}) => string | undefined;
|
|
271
|
-
}
|
|
272
|
-
/** Durable `<cell>/failure-receipt.json` written before a failed cell can
|
|
273
|
-
* trigger campaign-wide cancellation. The cell records dispatch measurements;
|
|
274
|
-
* `cost` covers every settled agent and judge call attributed to this exact run
|
|
275
|
-
* attempt. */
|
|
276
|
-
interface CampaignCellFailureReceipt<TArtifact = unknown> {
|
|
277
|
-
schemaVersion: 1;
|
|
278
|
-
runAttemptId: string;
|
|
279
|
-
recordedAt: string;
|
|
280
|
-
failure: {
|
|
281
|
-
stage: 'dispatch' | 'judge';
|
|
282
|
-
judge?: string;
|
|
283
|
-
error: {
|
|
284
|
-
name: string;
|
|
285
|
-
message: string;
|
|
286
|
-
stack?: string;
|
|
287
|
-
};
|
|
288
|
-
};
|
|
289
|
-
cell: CampaignCellResult<TArtifact>;
|
|
290
|
-
cost: CostLedgerSummary;
|
|
291
|
-
}
|
|
292
|
-
/**
|
|
293
|
-
* Bounded in-run retry of failed cells. Every attempt dispatches the same
|
|
294
|
-
* slot and charges the shared cost ledger, so the final cell's `costUsd`,
|
|
295
|
-
* `tokenUsage`, and `costCallIds` cover all attempts. Each retried attempt
|
|
296
|
-
* keeps its failure receipt at `<cell>/failure-receipt.attempt-<n>.json`; a
|
|
297
|
-
* final failed attempt keeps the usual `<cell>/failure-receipt.json`. The
|
|
298
|
-
* final cell records the retry count as `retryAttempts`. Artifacts and trace
|
|
299
|
-
* spans written by a later attempt replace those of the retried attempt; the
|
|
300
|
-
* per-attempt failure receipts are the durable evidence.
|
|
301
|
-
*/
|
|
302
|
-
interface CampaignCellRetryPolicy {
|
|
303
|
-
/** Total attempts per cell, including the first. A positive safe integer. */
|
|
304
|
-
attempts: number;
|
|
305
|
-
/** Decides whether a failed attempt is dispatched again. Receives the
|
|
306
|
-
* receipt's `failure` record. `transientDispatchFailure()` is the
|
|
307
|
-
* ready-made predicate for infrastructure hiccups (502/503/504, dropped
|
|
308
|
-
* streams, admission rejections). */
|
|
309
|
-
retryable: (failure: CampaignCellFailureReceipt['failure']) => boolean;
|
|
310
|
-
}
|
|
311
|
-
/**
|
|
312
|
-
* Core campaign orchestrator: fan scenarios through dispatch, score with judges, aggregate bootstrap CIs, and persist reproducible `CampaignResult` records.
|
|
313
|
-
*/
|
|
314
|
-
declare function runCampaign<TScenario extends Scenario, TArtifact>(opts: RunCampaignOptions<TScenario, TArtifact>): Promise<CampaignResult<TArtifact, TScenario>>;
|
|
315
|
-
//#endregion
|
|
316
|
-
//#region src/campaign/presets/run-eval.d.ts
|
|
317
|
-
interface RunEvalOptions<TScenario extends Scenario, TArtifact> extends Omit<RunCampaignOptions<TScenario, TArtifact>, 'runDir'> {
|
|
318
|
-
runDir: string;
|
|
319
|
-
}
|
|
320
|
-
/**
|
|
321
|
-
* Simplest evaluation preset: run scenarios through dispatch, score with judges, and return a `CampaignResult` — no optimizer, no gate, no PR.
|
|
322
|
-
*/
|
|
323
|
-
declare function runEval<TScenario extends Scenario, TArtifact>(opts: RunEvalOptions<TScenario, TArtifact>): Promise<CampaignResult<TArtifact, TScenario>>;
|
|
324
|
-
//#endregion
|
|
325
|
-
//#region src/campaign/auto-pr.d.ts
|
|
326
|
-
interface OpenAutoPrOptions<TArtifact, TScenario extends Scenario> {
|
|
327
|
-
/** Campaign result to attach to the PR. */
|
|
328
|
-
result: CampaignResult<TArtifact, TScenario>;
|
|
329
|
-
/** Gate verdict explaining the promotion. Substrate refuses to open a PR
|
|
330
|
-
* when `gate.decision !== 'ship'` — fails loud. */
|
|
331
|
-
gate: GateResult;
|
|
332
|
-
/** Promoted surface diff — typically the new system prompt addendum or
|
|
333
|
-
* full profile diff. Substrate writes it as the PR body. */
|
|
334
|
-
promotedDiff: string;
|
|
335
|
-
/** GH owner/repo target (e.g., `tangle-network/gtm-agent`). */
|
|
336
|
-
ghOwner: string;
|
|
337
|
-
ghRepo: string;
|
|
338
|
-
/** Branch name for the PR. Default `auto/<manifestHash[:12]>`. */
|
|
339
|
-
branch?: string;
|
|
340
|
-
/** PR title. Default includes manifest hash. */
|
|
341
|
-
title?: string;
|
|
342
|
-
/** Whether to actually open the PR or just dry-run. Default reads
|
|
343
|
-
* `GH_AUTO_PR_TOKEN` env — present = open, absent = dry-run. */
|
|
344
|
-
dryRun?: boolean;
|
|
345
|
-
/** Test seam — substitute `gh pr create` invocation. */
|
|
346
|
-
ghExec?: (args: string[]) => {
|
|
347
|
-
stdout: string;
|
|
348
|
-
stderr: string;
|
|
349
|
-
status: number;
|
|
350
|
-
};
|
|
351
|
-
}
|
|
352
|
-
interface OpenAutoPrResult {
|
|
353
|
-
opened: boolean;
|
|
354
|
-
prUrl?: string;
|
|
355
|
-
dryRun: boolean;
|
|
356
|
-
reason: string;
|
|
357
|
-
}
|
|
358
|
-
/**
|
|
359
|
-
* Open a GitHub PR for a gate-approved surface promotion, attaching the manifest hash, gate verdict, and diff as the PR body.
|
|
360
|
-
*/
|
|
361
|
-
declare function openAutoPr<TArtifact, TScenario extends Scenario>(options: OpenAutoPrOptions<TArtifact, TScenario>): OpenAutoPrResult;
|
|
362
|
-
//#endregion
|
|
363
|
-
//#region src/campaign/parent-selection.d.ts
|
|
364
|
-
/** Search state supplied to one parent-selection call. */
|
|
365
|
-
interface ParentSelectionContext {
|
|
366
|
-
/** Non-dominated scored surfaces across the whole run so far, including the
|
|
367
|
-
* baseline (`generation: -1`). Never empty. */
|
|
368
|
-
readonly frontier: ReadonlyArray<ParetoParent>;
|
|
369
|
-
/** Measured result of the global incumbent, the promotion bar. Under the
|
|
370
|
-
* default `selectionRankKey` the incumbent is always on `frontier`. */
|
|
371
|
-
readonly incumbent: ScoredSurfaceOutcome;
|
|
372
|
-
/** Every completed generation so far. */
|
|
373
|
-
readonly history: ReadonlyArray<GenerationRecord>;
|
|
374
|
-
/** Index of the generation about to propose. */
|
|
375
|
-
readonly generation: number;
|
|
376
|
-
}
|
|
377
|
-
/** Chooses the surface the next generation mutates. Returns one frontier
|
|
378
|
-
* parent; `runOptimization` refuses a parent it has not measured to
|
|
379
|
-
* completion or whose surface does not match its `surfaceHash`. */
|
|
380
|
-
type ParentSelector = (ctx: ParentSelectionContext) => ParetoParent;
|
|
381
|
-
interface CrowdedFrontierParentOptions {
|
|
382
|
-
/** Integer seed for the per-generation draw. The same seed, frontier, and
|
|
383
|
-
* generation index select the same parent. */
|
|
384
|
-
seed: number;
|
|
385
|
-
}
|
|
386
|
-
/**
|
|
387
|
-
* NSGA-II crowded tournament selection over the frontier. Each generation
|
|
388
|
-
* draws two distinct frontier members with a PRNG seeded from `seed` and the
|
|
389
|
-
* generation index, and keeps the one with the larger crowding distance (more
|
|
390
|
-
* isolated on the frontier). Boundary parents carry infinite distance, so a
|
|
391
|
-
* boundary parent always beats an interior one. A tie on distance falls back
|
|
392
|
-
* to the higher mean composite, then to the smaller surface hash. A frontier
|
|
393
|
-
* of one member returns that member.
|
|
394
|
-
*/
|
|
395
|
-
declare function crowdedFrontierParent(options: CrowdedFrontierParentOptions): ParentSelector;
|
|
396
|
-
//#endregion
|
|
397
|
-
//#region src/campaign/search-ledger.d.ts
|
|
398
|
-
declare const SEARCH_LEDGER_SCHEMA: 'tangle.search-ledger.v1';
|
|
399
|
-
type SearchLedgerHash = LedgerHash;
|
|
400
|
-
type SearchSurfaceKind = 'prompt' | 'tool-contract' | 'runtime-config' | 'memory' | 'knowledge' | 'agent-profile' | 'code' | 'deployment';
|
|
401
|
-
/** Content-addressed artifact or receipt. Mutable paths are locators only; the
|
|
402
|
-
* digest and byte length bind the exact bytes used by the search. */
|
|
403
|
-
interface SearchArtifactRef {
|
|
404
|
-
role: string;
|
|
405
|
-
uri: string;
|
|
406
|
-
sha256: SearchLedgerHash;
|
|
407
|
-
byteLength: number;
|
|
408
|
-
}
|
|
409
|
-
/** Repository, dataset, or package source pinned to an immutable commit or
|
|
410
|
-
* content digest. Branches, tags, and bare package versions are rejected. */
|
|
411
|
-
interface SearchSourceRef {
|
|
412
|
-
uri: string;
|
|
413
|
-
revision: string;
|
|
414
|
-
}
|
|
415
|
-
interface SearchModelIdentity {
|
|
416
|
-
provider: string;
|
|
417
|
-
snapshot: string;
|
|
418
|
-
}
|
|
419
|
-
interface SearchCandidateSurface {
|
|
420
|
-
surfaceId: string;
|
|
421
|
-
kind: SearchSurfaceKind;
|
|
422
|
-
artifact: SearchArtifactRef;
|
|
423
|
-
}
|
|
424
|
-
interface SearchCandidateLineage {
|
|
425
|
-
/** Existing `LineageNode.id`; this ledger references rather than embeds it. */
|
|
426
|
-
lineageNodeId: string;
|
|
427
|
-
parentCandidateIds: string[];
|
|
428
|
-
generation: number;
|
|
429
|
-
proposer: string;
|
|
430
|
-
proposerSource: SearchSourceRef;
|
|
431
|
-
}
|
|
432
|
-
type SearchOperationKind = 'candidate-generation' | 'analysis' | 'selection' | 'judge' | 'other';
|
|
433
|
-
interface SearchPlannedTask {
|
|
434
|
-
taskId: string;
|
|
435
|
-
source: SearchSourceRef;
|
|
436
|
-
benchmark: SearchSourceRef;
|
|
437
|
-
/** Maximum transport attempts for this task and candidate. Only an explicit
|
|
438
|
-
* passed/failed outcome satisfies the planned denominator. */
|
|
439
|
-
maxAttempts: number;
|
|
440
|
-
}
|
|
441
|
-
interface SearchPlannedOperation {
|
|
442
|
-
operationId: string;
|
|
443
|
-
kind: SearchOperationKind;
|
|
444
|
-
}
|
|
445
|
-
interface SearchCandidateSlot {
|
|
446
|
-
slotId: string;
|
|
447
|
-
/** Planned candidate-generation call that must either produce this slot or
|
|
448
|
-
* fail before the slot can be closed. Several slots may share one batched call. */
|
|
449
|
-
generationOperationId: string;
|
|
450
|
-
}
|
|
451
|
-
interface SearchPlan {
|
|
452
|
-
/** Stable slots and their proposer calls are frozen before search begins. */
|
|
453
|
-
candidateSlots: SearchCandidateSlot[];
|
|
454
|
-
/** Every task applies to every successfully registered candidate. */
|
|
455
|
-
tasks: SearchPlannedTask[];
|
|
456
|
-
/** Non-task spend slots: proposal, analysis, selection, extra judges, etc. */
|
|
457
|
-
operations: SearchPlannedOperation[];
|
|
458
|
-
}
|
|
459
|
-
type SearchTokenAccounting = {
|
|
460
|
-
status: 'known';
|
|
461
|
-
inputTokens: number;
|
|
462
|
-
outputTokens: number;
|
|
463
|
-
cachedTokens: number;
|
|
464
|
-
} | {
|
|
465
|
-
status: 'unknown';
|
|
466
|
-
reason: string;
|
|
467
|
-
};
|
|
468
|
-
type SearchCostAccounting = {
|
|
469
|
-
status: 'known';
|
|
470
|
-
usd: number;
|
|
471
|
-
source: 'provider' | 'pricing-table' | 'free';
|
|
472
|
-
} | {
|
|
473
|
-
status: 'unknown';
|
|
474
|
-
/** Known spend may still be a lower bound when one call was unpriced. */
|
|
475
|
-
knownLowerBoundUsd: number;
|
|
476
|
-
reason: string;
|
|
477
|
-
};
|
|
478
|
-
interface SearchAttemptAccounting {
|
|
479
|
-
tokens: SearchTokenAccounting;
|
|
480
|
-
cost: SearchCostAccounting;
|
|
481
|
-
}
|
|
482
|
-
interface SearchFailureReason {
|
|
483
|
-
code: string;
|
|
484
|
-
message: string;
|
|
485
|
-
}
|
|
486
|
-
type SearchTaskOutcome = {
|
|
487
|
-
status: 'passed';
|
|
488
|
-
score: number;
|
|
489
|
-
metrics: Record<string, number>;
|
|
490
|
-
} | {
|
|
491
|
-
status: 'failed';
|
|
492
|
-
score: number;
|
|
493
|
-
metrics: Record<string, number>;
|
|
494
|
-
failure: SearchFailureReason;
|
|
495
|
-
} | {
|
|
496
|
-
status: 'errored';
|
|
497
|
-
metrics: Record<string, number>;
|
|
498
|
-
error: SearchFailureReason & {
|
|
499
|
-
retryable: boolean;
|
|
500
|
-
};
|
|
501
|
-
};
|
|
502
|
-
type SearchSurfaceEffect = {
|
|
503
|
-
status: 'measured';
|
|
504
|
-
metric: string;
|
|
505
|
-
baselineValue: number;
|
|
506
|
-
candidateValue: number;
|
|
507
|
-
delta: number;
|
|
508
|
-
} | {
|
|
509
|
-
status: 'not-measured';
|
|
510
|
-
reason: string;
|
|
511
|
-
};
|
|
512
|
-
/** Per-attempt proof that a declared candidate surface was or was not active,
|
|
513
|
-
* plus measured effect when the experiment supports attribution. */
|
|
514
|
-
interface SearchSurfaceEvidence {
|
|
515
|
-
surfaceId: string;
|
|
516
|
-
fired: boolean;
|
|
517
|
-
firingCount: number;
|
|
518
|
-
effect: SearchSurfaceEffect;
|
|
519
|
-
evidence: SearchArtifactRef[];
|
|
520
|
-
}
|
|
521
|
-
interface SearchLedgerEventBase {
|
|
522
|
-
eventId: string;
|
|
523
|
-
occurredAt: string;
|
|
524
|
-
artifacts: SearchArtifactRef[];
|
|
525
|
-
}
|
|
526
|
-
interface SearchPlannedEvent extends SearchLedgerEventBase {
|
|
527
|
-
kind: 'search-planned';
|
|
528
|
-
plan: SearchPlan;
|
|
529
|
-
}
|
|
530
|
-
/** Additional candidate slots and operations for a search whose length is not
|
|
531
|
-
* known when it starts. The plan stays the first event and the planned task
|
|
532
|
-
* denominator stays frozen: extending tasks would retroactively reopen
|
|
533
|
-
* candidates that already closed theirs. */
|
|
534
|
-
interface SearchPlanExtendedEvent extends SearchLedgerEventBase {
|
|
535
|
-
kind: 'search-plan-extended';
|
|
536
|
-
extension: {
|
|
537
|
-
candidateSlots: SearchCandidateSlot[];
|
|
538
|
-
operations: SearchPlannedOperation[];
|
|
539
|
-
};
|
|
540
|
-
}
|
|
541
|
-
interface SearchCandidateRegisteredEvent extends SearchLedgerEventBase {
|
|
542
|
-
kind: 'candidate-registered';
|
|
543
|
-
slotId: string;
|
|
544
|
-
generationOperationId: string;
|
|
545
|
-
candidateId: string;
|
|
546
|
-
lineage: SearchCandidateLineage;
|
|
547
|
-
surfaces: SearchCandidateSurface[];
|
|
548
|
-
}
|
|
549
|
-
interface SearchCandidateSlotClosedEvent extends SearchLedgerEventBase {
|
|
550
|
-
kind: 'candidate-slot-closed';
|
|
551
|
-
slotId: string;
|
|
552
|
-
generationOperationId: string;
|
|
553
|
-
reason: SearchFailureReason;
|
|
554
|
-
}
|
|
555
|
-
interface SearchTaskAttemptedEvent extends SearchLedgerEventBase {
|
|
556
|
-
kind: 'task-attempted';
|
|
557
|
-
candidateId: string;
|
|
558
|
-
runId: string;
|
|
559
|
-
attemptIndex: number;
|
|
560
|
-
task: {
|
|
561
|
-
taskId: string;
|
|
562
|
-
source: SearchSourceRef;
|
|
563
|
-
};
|
|
564
|
-
identity: {
|
|
565
|
-
model: SearchModelIdentity;
|
|
566
|
-
agent: SearchSourceRef;
|
|
567
|
-
benchmark: SearchSourceRef;
|
|
568
|
-
};
|
|
569
|
-
outcome: SearchTaskOutcome;
|
|
570
|
-
accounting: SearchAttemptAccounting;
|
|
571
|
-
surfaceEvidence: SearchSurfaceEvidence[];
|
|
572
|
-
}
|
|
573
|
-
interface SearchOperationRecordedEvent extends SearchLedgerEventBase {
|
|
574
|
-
kind: 'search-operation-recorded';
|
|
575
|
-
operationId: string;
|
|
576
|
-
operationKind: SearchOperationKind;
|
|
577
|
-
execution: {
|
|
578
|
-
kind: 'model';
|
|
579
|
-
model: SearchModelIdentity;
|
|
580
|
-
source: SearchSourceRef;
|
|
581
|
-
} | {
|
|
582
|
-
kind: 'deterministic';
|
|
583
|
-
source: SearchSourceRef;
|
|
584
|
-
};
|
|
585
|
-
outcome: {
|
|
586
|
-
status: 'completed';
|
|
587
|
-
} | {
|
|
588
|
-
status: 'partial';
|
|
589
|
-
failure: SearchFailureReason;
|
|
590
|
-
} | {
|
|
591
|
-
status: 'failed';
|
|
592
|
-
failure: SearchFailureReason;
|
|
593
|
-
};
|
|
594
|
-
accounting: SearchAttemptAccounting;
|
|
595
|
-
}
|
|
596
|
-
interface SearchCandidateDecidedEvent extends SearchLedgerEventBase {
|
|
597
|
-
kind: 'candidate-decided';
|
|
598
|
-
candidateId: string;
|
|
599
|
-
decision: {
|
|
600
|
-
status: 'selected';
|
|
601
|
-
} | {
|
|
602
|
-
status: 'rejected';
|
|
603
|
-
reason: SearchFailureReason;
|
|
604
|
-
};
|
|
605
|
-
}
|
|
606
|
-
interface SearchCompletedEvent extends SearchLedgerEventBase {
|
|
607
|
-
kind: 'search-completed';
|
|
608
|
-
result: {
|
|
609
|
-
status: 'selected';
|
|
610
|
-
candidateId: string;
|
|
611
|
-
} | {
|
|
612
|
-
status: 'all-rejected';
|
|
613
|
-
reason: SearchFailureReason;
|
|
614
|
-
};
|
|
615
|
-
}
|
|
616
|
-
type SearchLedgerEvent = SearchPlannedEvent | SearchPlanExtendedEvent | SearchCandidateRegisteredEvent | SearchCandidateSlotClosedEvent | SearchTaskAttemptedEvent | SearchOperationRecordedEvent | SearchCandidateDecidedEvent | SearchCompletedEvent;
|
|
617
|
-
interface SearchLedgerEntry {
|
|
618
|
-
schema: typeof SEARCH_LEDGER_SCHEMA;
|
|
619
|
-
campaignId: string;
|
|
620
|
-
sequence: number;
|
|
621
|
-
previousHash: SearchLedgerHash | null;
|
|
622
|
-
event: SearchLedgerEvent;
|
|
623
|
-
entryHash: SearchLedgerHash;
|
|
624
|
-
}
|
|
625
|
-
type SearchAccountingAudit = {
|
|
626
|
-
status: 'known';
|
|
627
|
-
inputTokens: number;
|
|
628
|
-
outputTokens: number;
|
|
629
|
-
cachedTokens: number;
|
|
630
|
-
costUsd: number;
|
|
631
|
-
} | {
|
|
632
|
-
status: 'partial';
|
|
633
|
-
knownInputTokens: number;
|
|
634
|
-
knownOutputTokens: number;
|
|
635
|
-
knownCachedTokens: number;
|
|
636
|
-
knownCostUsd: number;
|
|
637
|
-
unknownTokenEventIds: string[];
|
|
638
|
-
unknownCostEventIds: string[];
|
|
639
|
-
};
|
|
640
|
-
interface SearchLedgerAudit {
|
|
641
|
-
campaignId: string;
|
|
642
|
-
eventCount: number;
|
|
643
|
-
candidateCount: number;
|
|
644
|
-
closedCandidateSlotCount: number;
|
|
645
|
-
attemptCount: number;
|
|
646
|
-
operationCount: number;
|
|
647
|
-
outcomes: {
|
|
648
|
-
passed: number;
|
|
649
|
-
failed: number;
|
|
650
|
-
errored: number;
|
|
651
|
-
};
|
|
652
|
-
operationOutcomes: {
|
|
653
|
-
completed: number;
|
|
654
|
-
partial: number;
|
|
655
|
-
failed: number;
|
|
656
|
-
};
|
|
657
|
-
decisions: {
|
|
658
|
-
selected: number;
|
|
659
|
-
rejected: number;
|
|
660
|
-
pending: number;
|
|
661
|
-
};
|
|
662
|
-
expected: {
|
|
663
|
-
candidateSlots: number;
|
|
664
|
-
taskOutcomes: number;
|
|
665
|
-
operations: number;
|
|
666
|
-
missingCandidateSlots: string[];
|
|
667
|
-
missingTaskOutcomes: string[];
|
|
668
|
-
missingOperations: string[];
|
|
669
|
-
};
|
|
670
|
-
status: 'in-progress' | 'selected' | 'all-rejected';
|
|
671
|
-
selectedCandidateId: string | null;
|
|
672
|
-
accounting: SearchAccountingAudit;
|
|
673
|
-
headHash: SearchLedgerHash | null;
|
|
674
|
-
}
|
|
675
|
-
interface SearchLedgerReplay {
|
|
676
|
-
entries: SearchLedgerEntry[];
|
|
677
|
-
plan: SearchPlannedEvent | null;
|
|
678
|
-
/** Appended plan extensions, in ledger order. The effective plan is the
|
|
679
|
-
* first plan event merged with these; `audit.expected` counts the merge. */
|
|
680
|
-
planExtensions: SearchPlanExtendedEvent[];
|
|
681
|
-
candidates: SearchCandidateRegisteredEvent[];
|
|
682
|
-
closedCandidateSlots: SearchCandidateSlotClosedEvent[];
|
|
683
|
-
attempts: SearchTaskAttemptedEvent[];
|
|
684
|
-
operations: SearchOperationRecordedEvent[];
|
|
685
|
-
decisions: SearchCandidateDecidedEvent[];
|
|
686
|
-
completion: SearchCompletedEvent | null;
|
|
687
|
-
audit: SearchLedgerAudit;
|
|
688
|
-
}
|
|
689
|
-
interface SearchLedgerAppendResult {
|
|
690
|
-
entry: SearchLedgerEntry;
|
|
691
|
-
/** False when the exact event was already durably present. */
|
|
692
|
-
appended: boolean;
|
|
693
|
-
replay: SearchLedgerReplay;
|
|
694
|
-
}
|
|
695
|
-
/** Validate and return a canonical copy. Arrays whose order is not semantic are
|
|
696
|
-
* sorted so retries from different processes produce byte-identical events. */
|
|
697
|
-
declare function validateSearchLedgerEvent(input: unknown): SearchLedgerEvent;
|
|
698
|
-
/**
|
|
699
|
-
* How this ledger uses its trusted head — the `(sequence, entryHash)` pin kept
|
|
700
|
-
* in the sibling `<path>.head` file that a hash chain needs to prove entries
|
|
701
|
-
* were not deleted from the end. `ledger-core/trusted-head.ts` holds the threat
|
|
702
|
-
* model.
|
|
703
|
-
*
|
|
704
|
-
* - `pin` (default): every append records the new head, and a pin that is
|
|
705
|
-
* present is verified on every read.
|
|
706
|
-
* - `require`: additionally refuses to read a non-empty ledger whose pin is
|
|
707
|
-
* gone, so deleting the sibling file cannot downgrade the guarantee. Only for
|
|
708
|
-
* ledgers written under `pin` from their first entry.
|
|
709
|
-
* - `off`: chain verification only. Truncation to a valid shorter prefix is
|
|
710
|
-
* undetectable.
|
|
711
|
-
*/
|
|
712
|
-
type SearchLedgerTrustedHeadMode = 'pin' | 'require' | 'off';
|
|
713
|
-
interface OpenSearchLedgerOptions {
|
|
714
|
-
path: string;
|
|
715
|
-
campaignId: string;
|
|
716
|
-
trustedHead?: SearchLedgerTrustedHeadMode;
|
|
717
|
-
}
|
|
718
|
-
interface SearchLedger {
|
|
719
|
-
readonly path: string;
|
|
720
|
-
readonly campaignId: string;
|
|
721
|
-
/** Sibling file holding this ledger's trusted head. */
|
|
722
|
-
readonly trustedHeadPath: string;
|
|
723
|
-
append(event: SearchLedgerEvent): Promise<SearchLedgerAppendResult>;
|
|
724
|
-
replay(): Promise<SearchLedgerReplay>;
|
|
725
|
-
/** The pinned head, or null when this ledger has never been pinned. */
|
|
726
|
-
trustedHead(): Promise<LedgerTrustedHead | null>;
|
|
727
|
-
/** Pin the current verified head: how a ledger written under `off`, or one
|
|
728
|
-
* whose pin file was removed, acquires a pin without rewriting a byte. */
|
|
729
|
-
pinTrustedHead(): Promise<LedgerTrustedHead>;
|
|
730
|
-
/** Discard this ledger's pin, reporting what was discarded. Deleting or
|
|
731
|
-
* rebuilding the ledger file leaves a pin naming history the file no longer
|
|
732
|
-
* carries, and every later read is refused because that is exactly the
|
|
733
|
-
* deletion the pin exists to catch; clearing is the supported way to abandon
|
|
734
|
-
* that history on purpose. It gives up the deletion guarantee for every entry
|
|
735
|
-
* the pin covered. */
|
|
736
|
-
clearTrustedHead(): Promise<LedgerTrustedHeadRemoval>;
|
|
737
|
-
}
|
|
738
|
-
/** Open a durable filesystem search ledger. Construction performs no I/O; the
|
|
739
|
-
* first `append` or `replay` validates the complete existing file. */
|
|
740
|
-
declare function openSearchLedger(options: OpenSearchLedgerOptions): SearchLedger;
|
|
741
|
-
/** Append-only file-backed search ledger with idempotent writes and replay. */
|
|
742
|
-
declare class FileSearchLedger implements SearchLedger {
|
|
743
|
-
readonly path: string;
|
|
744
|
-
readonly campaignId: string;
|
|
745
|
-
readonly trustedHeadPath: string;
|
|
746
|
-
private readonly trustedHeadMode;
|
|
747
|
-
private readonly journal;
|
|
748
|
-
constructor(path: string, campaignId: string, trustedHead?: SearchLedgerTrustedHeadMode);
|
|
749
|
-
replay(): Promise<SearchLedgerReplay>;
|
|
750
|
-
append(input: SearchLedgerEvent): Promise<SearchLedgerAppendResult>;
|
|
751
|
-
trustedHead(): Promise<LedgerTrustedHead | null>;
|
|
752
|
-
pinTrustedHead(): Promise<LedgerTrustedHead>;
|
|
753
|
-
clearTrustedHead(): Promise<LedgerTrustedHeadRemoval>;
|
|
754
|
-
}
|
|
755
|
-
//#endregion
|
|
756
|
-
//#region src/campaign/search-history-receipt.d.ts
|
|
757
|
-
declare const SEARCH_HISTORY_RECEIPT_SCHEMA_VERSION: '1.0.0';
|
|
758
|
-
declare const SEARCH_HISTORY_RECEIPT_DIGEST_ALGORITHM: 'rfc8785-sha256';
|
|
759
|
-
/** Bounded projection of the canonical replay audit. Exact ids stay in SearchLedger. */
|
|
760
|
-
interface SearchHistoryAuditSummary {
|
|
761
|
-
readonly campaignId: string;
|
|
762
|
-
readonly headHash: SearchLedgerHash | null;
|
|
763
|
-
readonly status: SearchLedgerAudit['status'];
|
|
764
|
-
readonly selectedCandidateId: string | null;
|
|
765
|
-
readonly eventCount: number;
|
|
766
|
-
readonly candidateCount: number;
|
|
767
|
-
readonly closedCandidateSlotCount: number;
|
|
768
|
-
readonly attemptCount: number;
|
|
769
|
-
readonly operationCount: number;
|
|
770
|
-
readonly expectedCandidateSlots: number;
|
|
771
|
-
readonly expectedTaskOutcomes: number;
|
|
772
|
-
readonly expectedOperations: number;
|
|
773
|
-
readonly missingCandidateSlots: number;
|
|
774
|
-
readonly missingTaskOutcomes: number;
|
|
775
|
-
readonly missingOperations: number;
|
|
776
|
-
readonly pendingDecisions: number;
|
|
777
|
-
readonly hasPlan: boolean;
|
|
778
|
-
readonly hasCompletion: boolean;
|
|
779
|
-
}
|
|
780
|
-
/**
|
|
781
|
-
* A bounded proof envelope over one canonical SearchLedger replay.
|
|
782
|
-
*
|
|
783
|
-
* The content-addressed ledger remains the sole rich history. This receipt binds
|
|
784
|
-
* its producer/run identity, exact audit digest, bounded completeness summary,
|
|
785
|
-
* and its own canonical digest. Consumers needing candidate ids, attempts,
|
|
786
|
-
* failures, accounting gaps, or decisions read and replay the canonical ledger.
|
|
787
|
-
*/
|
|
788
|
-
interface SearchHistoryReceipt {
|
|
789
|
-
readonly schemaVersion: typeof SEARCH_HISTORY_RECEIPT_SCHEMA_VERSION;
|
|
790
|
-
readonly kind: 'search-history-receipt';
|
|
791
|
-
readonly digestAlgorithm: typeof SEARCH_HISTORY_RECEIPT_DIGEST_ALGORITHM;
|
|
792
|
-
readonly receiptDigest: SearchLedgerHash;
|
|
793
|
-
/** Stable producer identity, for example an OptimizationMethod name. */
|
|
794
|
-
readonly producerId: string;
|
|
795
|
-
/** Concrete optimizer/runtime invocation that produced the ledger. */
|
|
796
|
-
readonly runId: string;
|
|
797
|
-
readonly ledger: SearchArtifactRef;
|
|
798
|
-
/** Digest of the complete SearchLedgerAudit produced by canonical replay. */
|
|
799
|
-
readonly auditDigest: SearchLedgerHash;
|
|
800
|
-
readonly summary: SearchHistoryAuditSummary;
|
|
801
|
-
readonly complete: boolean;
|
|
802
|
-
readonly incompleteReasons: readonly string[];
|
|
803
|
-
}
|
|
804
|
-
interface CreateSearchHistoryReceiptInput {
|
|
805
|
-
readonly producerId: string;
|
|
806
|
-
readonly runId: string;
|
|
807
|
-
/** Content-addressed canonical SearchLedger JSONL artifact. */
|
|
808
|
-
readonly ledger: SearchArtifactRef;
|
|
809
|
-
/** The result returned by SearchLedger.replay(). */
|
|
810
|
-
readonly replay: SearchLedgerReplay;
|
|
811
|
-
}
|
|
812
|
-
type SearchHistoryPolicy = 'allow-missing' | 'require-complete';
|
|
813
|
-
interface SearchHistoryCoverageRow {
|
|
814
|
-
readonly producerId: string;
|
|
815
|
-
readonly status: 'complete' | 'incomplete' | 'missing';
|
|
816
|
-
readonly reasons: readonly string[];
|
|
817
|
-
readonly receipt?: SearchHistoryReceipt;
|
|
818
|
-
}
|
|
819
|
-
interface SearchHistoryCoverage {
|
|
820
|
-
readonly policy: SearchHistoryPolicy;
|
|
821
|
-
readonly allComplete: boolean;
|
|
822
|
-
readonly producers: readonly SearchHistoryCoverageRow[];
|
|
823
|
-
}
|
|
824
|
-
declare class SearchHistoryRequiredError extends Error {
|
|
825
|
-
readonly producerId: string;
|
|
826
|
-
readonly reasons: readonly string[];
|
|
827
|
-
constructor(producerId: string, reasons: readonly string[]);
|
|
828
|
-
}
|
|
829
|
-
/** Build a bounded receipt from the projection returned by canonical ledger replay. */
|
|
830
|
-
declare function createSearchHistoryReceipt(input: CreateSearchHistoryReceiptInput): SearchHistoryReceipt;
|
|
831
|
-
/** Verify the bounded receipt. Full ledger bytes are verified by SearchLedger. */
|
|
832
|
-
declare function verifySearchHistoryReceipt(receipt: SearchHistoryReceipt): SearchHistoryReceipt;
|
|
833
|
-
/** Prove that a receipt still describes the exact canonical replay supplied. */
|
|
834
|
-
declare function assertSearchHistoryMatchesReplay(receipt: SearchHistoryReceipt, replay: SearchLedgerReplay): void;
|
|
835
|
-
/** Require a receipt owned by this producer and a terminal, denominator-complete history. */
|
|
836
|
-
declare function assertCompleteSearchHistory(producerId: string, receipt: SearchHistoryReceipt | undefined): asserts receipt is SearchHistoryReceipt;
|
|
837
|
-
/** Classify one producer's history without treating malformed evidence as absence. */
|
|
838
|
-
declare function searchHistoryCoverageRow(producerId: string, receipt: SearchHistoryReceipt | undefined): SearchHistoryCoverageRow;
|
|
839
|
-
//#endregion
|
|
840
|
-
//#region src/campaign/gepa-candidate-population.d.ts
|
|
841
|
-
interface GepaCandidatePopulationSummary {
|
|
842
|
-
readonly scope: 'gepa-candidate-population';
|
|
843
|
-
readonly path: string;
|
|
844
|
-
readonly sha256: `sha256:${string}`;
|
|
845
|
-
readonly bytes: number;
|
|
846
|
-
readonly runId: string;
|
|
847
|
-
readonly candidates: number;
|
|
848
|
-
readonly bestIndex: number;
|
|
849
|
-
readonly maxCandidates: number;
|
|
850
|
-
readonly maxCandidateChars: number;
|
|
851
|
-
readonly scenarioIds: readonly string[];
|
|
852
|
-
readonly surfaceKind: 'text' | 'components';
|
|
853
|
-
}
|
|
854
|
-
interface GepaCandidateSelectionScore {
|
|
855
|
-
readonly scenarioId: string;
|
|
856
|
-
readonly score: number;
|
|
857
|
-
}
|
|
858
|
-
interface GepaCandidatePopulationCandidate {
|
|
859
|
-
/** Zero-based index assigned by the exact GEPA result. */
|
|
860
|
-
readonly index: number;
|
|
861
|
-
readonly candidate: ExternalTextCandidate;
|
|
862
|
-
readonly candidateHash: string;
|
|
863
|
-
readonly candidateDigest: `sha256:${string}`;
|
|
864
|
-
/** Exact GEPA parent indices. The seed candidate has one null parent. */
|
|
865
|
-
readonly parentIndices: readonly (number | null)[];
|
|
866
|
-
/** Null means GEPA had no selection score for this candidate. */
|
|
867
|
-
readonly aggregateScore: number | null;
|
|
868
|
-
readonly selectionScores: readonly GepaCandidateSelectionScore[];
|
|
869
|
-
readonly discoveryEvaluationCount: number;
|
|
870
|
-
}
|
|
871
|
-
interface GepaCandidatePopulationArtifact {
|
|
872
|
-
readonly summary: GepaCandidatePopulationSummary;
|
|
873
|
-
readonly runId: string;
|
|
874
|
-
readonly bestIndex: number;
|
|
875
|
-
readonly candidates: readonly GepaCandidatePopulationCandidate[];
|
|
876
|
-
}
|
|
877
|
-
/**
|
|
878
|
-
* Read GEPA's exact candidate graph from the artifact addressed by method provenance.
|
|
879
|
-
*
|
|
880
|
-
* The reader checks the supplied digest, declared byte count, run identity,
|
|
881
|
-
* candidate surfaces, parent graph, selection scores, and configured bounds.
|
|
882
|
-
* This proves that the bytes match the supplied summary. The caller remains
|
|
883
|
-
* responsible for obtaining that summary from trusted method provenance.
|
|
884
|
-
*/
|
|
885
|
-
declare function readGepaCandidatePopulationArtifact(input: {
|
|
886
|
-
summary: GepaCandidatePopulationSummary;
|
|
887
|
-
storage?: CampaignStorage;
|
|
888
|
-
}): GepaCandidatePopulationArtifact;
|
|
889
|
-
//#endregion
|
|
890
|
-
//#region src/campaign/search-ledger-recording.d.ts
|
|
891
|
-
/** How a search operation executed. The shape the ledger event records. */
|
|
892
|
-
type SearchExecutionIdentity = SearchOperationRecordedEvent['execution'];
|
|
893
|
-
/** Immutable identities the ledger requires and a campaign cannot infer. */
|
|
894
|
-
interface SearchRunIdentity {
|
|
895
|
-
/** The agent implementation under optimization. */
|
|
896
|
-
agent: SearchSourceRef;
|
|
897
|
-
/** The candidate generator: a model call or deterministic code. */
|
|
898
|
-
proposer: SearchExecutionIdentity;
|
|
899
|
-
/** The code that plans the search and selects its winner. */
|
|
900
|
-
search: SearchSourceRef;
|
|
901
|
-
/** Model the agent runs. Used only for a cell that reported none. */
|
|
902
|
-
model: SearchModelIdentity;
|
|
903
|
-
}
|
|
904
|
-
interface SearchLedgerBinding {
|
|
905
|
-
ledger: SearchLedger;
|
|
906
|
-
identity: SearchRunIdentity;
|
|
907
|
-
}
|
|
908
|
-
/** One proposed candidate, before it is measured. */
|
|
909
|
-
interface ProposedSearchCandidate {
|
|
910
|
-
surface: MutableSurface;
|
|
911
|
-
surfaceHash: string;
|
|
912
|
-
label?: string;
|
|
913
|
-
}
|
|
914
|
-
/** One measured candidate, after its campaign scored. */
|
|
915
|
-
interface MeasuredSearchCandidate<TArtifact> {
|
|
916
|
-
surface: MutableSurface;
|
|
917
|
-
surfaceHash: string;
|
|
918
|
-
cells: ReadonlyArray<CampaignCellResult<TArtifact>>;
|
|
919
|
-
runDir: string;
|
|
920
|
-
/** False when the candidate missed a designed cell. */
|
|
921
|
-
coverageComplete: boolean;
|
|
922
|
-
}
|
|
923
|
-
interface SearchRecorderOptions<TScenario extends Scenario> {
|
|
924
|
-
binding: SearchLedgerBinding;
|
|
925
|
-
storage: CampaignStorage;
|
|
926
|
-
runDir: string;
|
|
927
|
-
scenarios: ReadonlyArray<TScenario>;
|
|
928
|
-
reps: number;
|
|
929
|
-
maxGenerations: number;
|
|
930
|
-
populationSize: number;
|
|
931
|
-
/** Identity of the exact campaign design; the task benchmark pin. */
|
|
932
|
-
splitDigest: `sha256:${string}`;
|
|
933
|
-
/** Proposer label recorded on every candidate lineage. */
|
|
934
|
-
proposerLabel: string;
|
|
935
|
-
costLedger: CostLedgerHandle;
|
|
936
|
-
}
|
|
937
|
-
/**
|
|
938
|
-
* Recorder for one `runOptimization` run: `open()`, then `recordGeneration()`
|
|
939
|
-
* and `recordResults()` per generation, then `finish()`.
|
|
940
|
-
*
|
|
941
|
-
* Every event id is derived from the run, and an id already durable is not
|
|
942
|
-
* appended again, so a resumed run continues one ledger instead of conflicting
|
|
943
|
-
* with its own history.
|
|
944
|
-
*/
|
|
945
|
-
declare class SearchRecorder<TScenario extends Scenario, TArtifact> {
|
|
946
|
-
private readonly opts;
|
|
947
|
-
private readonly tasks;
|
|
948
|
-
private readonly registered;
|
|
949
|
-
private readonly order;
|
|
950
|
-
private readonly coverage;
|
|
951
|
-
private readonly openSlots;
|
|
952
|
-
private readonly durableEventIds;
|
|
953
|
-
private lastStampMs;
|
|
954
|
-
private proposalReceiptCount;
|
|
955
|
-
private constructor();
|
|
956
|
-
/** Open the recorder and append the plan. An existing ledger for the same
|
|
957
|
-
* run is re-read first, so a resumed run keeps one plan and one lineage. */
|
|
958
|
-
static open<TScenario extends Scenario, TArtifact>(opts: SearchRecorderOptions<TScenario>): Promise<SearchRecorder<TScenario, TArtifact>>;
|
|
959
|
-
/**
|
|
960
|
-
* Record one generation's candidate-generation call and the candidates it
|
|
961
|
-
* produced. A proposal larger than the planned population extends the plan
|
|
962
|
-
* with the extra slots; a proposal that fills fewer closes the rest.
|
|
963
|
-
*/
|
|
964
|
-
recordGeneration(input: {
|
|
965
|
-
generation: number;
|
|
966
|
-
parentSurfaceHash: string;
|
|
967
|
-
candidates: ReadonlyArray<ProposedSearchCandidate>;
|
|
968
|
-
}): Promise<void>;
|
|
969
|
-
/** Append one task attempt per designed cell of each candidate campaign. */
|
|
970
|
-
recordResults(candidates: ReadonlyArray<MeasuredSearchCandidate<TArtifact>>): Promise<void>;
|
|
971
|
-
/**
|
|
972
|
-
* Close the search: unreached generations, the selection operation, one
|
|
973
|
-
* decision per candidate, then the terminal event.
|
|
974
|
-
*
|
|
975
|
-
* The terminal event is appended only when canonical replay accounts for the
|
|
976
|
-
* whole planned denominator. An interrupted or partly unscored search stays
|
|
977
|
-
* `in-progress` and its receipt reports the exact gap, instead of claiming a
|
|
978
|
-
* closed search.
|
|
979
|
-
*/
|
|
980
|
-
finish(input: {
|
|
981
|
-
winnerSurfaceHash: string;
|
|
982
|
-
generationsRun: number;
|
|
983
|
-
runId: string;
|
|
984
|
-
}): Promise<SearchHistoryReceipt>;
|
|
985
|
-
/** Bounded receipt over the exact ledger bytes this run produced. */
|
|
986
|
-
receipt(runId: string): Promise<SearchHistoryReceipt>;
|
|
987
|
-
/** Read an existing ledger for this run so a resume continues it. */
|
|
988
|
-
private hydrate;
|
|
989
|
-
private plan;
|
|
990
|
-
private registeredSlot;
|
|
991
|
-
private remember;
|
|
992
|
-
private recordGenerationOperation;
|
|
993
|
-
private closeSlot;
|
|
994
|
-
/** Spend booked to candidate generation since the previous generation. */
|
|
995
|
-
private proposalAccounting;
|
|
996
|
-
private cellModel;
|
|
997
|
-
private proposalArtifact;
|
|
998
|
-
/** Write one canonical evidence document and return its content address. */
|
|
999
|
-
private writeArtifact;
|
|
1000
|
-
private append;
|
|
1001
|
-
/** Non-decreasing ISO stamps; the ledger refuses an event that moves back. */
|
|
1002
|
-
private stamp;
|
|
1003
|
-
}
|
|
1004
|
-
/**
|
|
1005
|
-
* Record an optimizer's own candidate graph into the same ledger.
|
|
1006
|
-
*
|
|
1007
|
-
* A complete optimization method searches inside its own process and reports
|
|
1008
|
-
* one artifact when it finishes: the candidate population, with each
|
|
1009
|
-
* candidate's parents and its score per selection scenario. This turns that
|
|
1010
|
-
* artifact into the canonical event stream, so a first-party method returns
|
|
1011
|
-
* the same `SearchHistoryReceipt` the in-process loop returns, and
|
|
1012
|
-
* `compareOptimizationMethods({ searchHistoryPolicy: 'require-complete' })`
|
|
1013
|
-
* accepts it.
|
|
1014
|
-
*
|
|
1015
|
-
* A candidate the optimizer left unscored on a planned scenario leaves the
|
|
1016
|
-
* planned denominator open, so the receipt reports the gap instead of closing
|
|
1017
|
-
* the search.
|
|
1018
|
-
*/
|
|
1019
|
-
declare function recordCandidatePopulationSearch<TScenario extends Scenario>(input: {
|
|
1020
|
-
ledger: SearchLedger;
|
|
1021
|
-
storage: CampaignStorage;
|
|
1022
|
-
runDir: string;
|
|
1023
|
-
identity: SearchRunIdentity;
|
|
1024
|
-
population: GepaCandidatePopulationArtifact;
|
|
1025
|
-
/** Scenarios the optimizer selected on. Must cover the population's ids. */
|
|
1026
|
-
scenarios: ReadonlyArray<TScenario>;
|
|
1027
|
-
/** Spend the optimizer booked to its own candidate generation. */
|
|
1028
|
-
generationAccounting: SearchAttemptAccounting;
|
|
1029
|
-
producerId: string;
|
|
1030
|
-
runId: string;
|
|
1031
|
-
}): Promise<SearchHistoryReceipt>;
|
|
1032
|
-
//#endregion
|
|
1033
|
-
//#region src/campaign/presets/run-optimization.d.ts
|
|
1034
|
-
interface PremeasuredOptimizationBaseline<TArtifact, TScenario extends Scenario> {
|
|
1035
|
-
/** Hash of the exact surface that produced `campaign`. */
|
|
1036
|
-
surfaceHash: string;
|
|
1037
|
-
/** Complete prior measurement reused by identity, including artifactsByPath. */
|
|
1038
|
-
campaign: CampaignResult<TArtifact, TScenario>;
|
|
1039
|
-
}
|
|
1040
|
-
interface RunOptimizationBaseOptions<TScenario extends Scenario, TArtifact> extends Omit<RunCampaignOptions<TScenario, TArtifact>, 'dispatch'> {
|
|
1041
|
-
/** Initial mutable surface (typically system prompt or addendum). */
|
|
1042
|
-
baselineSurface: MutableSurface;
|
|
1043
|
-
/**
|
|
1044
|
-
* Complete prior measurement of `baselineSurface`. When present,
|
|
1045
|
-
* `runOptimization` validates its surface, scenario split, seed, reps, and
|
|
1046
|
-
* normal campaign coverage, then skips the baseline campaign entirely — no
|
|
1047
|
-
* dispatch or resumability-cache lookup. Candidate campaigns still run
|
|
1048
|
-
* normally. Prior spend remains in the imported campaign aggregates and is
|
|
1049
|
-
* not added again to this continuation's CostLedger.
|
|
1050
|
-
*/
|
|
1051
|
-
premeasuredBaseline?: PremeasuredOptimizationBaseline<TArtifact, TScenario>;
|
|
1052
|
-
/** Dispatcher that takes the CURRENT surface + scenario → artifact. */
|
|
1053
|
-
dispatchWithSurface: (surface: MutableSurface, scenario: TScenario, ctx: Parameters<RunCampaignOptions<TScenario, TArtifact>['dispatch']>[1]) => Promise<TArtifact>;
|
|
1054
|
-
/** The candidate-generation strategy. */
|
|
1055
|
-
proposer: SurfaceProposer<ProposalFinding>;
|
|
1056
|
-
populationSize: number;
|
|
1057
|
-
maxGenerations: number;
|
|
1058
|
-
/** Candidate campaigns run at once. Default 1. Total concurrent cells are
|
|
1059
|
-
* bounded by candidateConcurrency * maxConcurrency. */
|
|
1060
|
-
candidateConcurrency?: number;
|
|
1061
|
-
/** DEPTH knob forwarded to the proposer's `propose()` — max iterations the
|
|
1062
|
-
* agentic generator may take per candidate. */
|
|
1063
|
-
maxImprovementShots?: number;
|
|
1064
|
-
/** Search or observed-production findings forwarded to candidate generation. */
|
|
1065
|
-
findings?: ReadonlyArray<ProposalFinding>;
|
|
1066
|
-
/** Per-generation findings producer. Runs once on the BASELINE campaign
|
|
1067
|
-
* (as `generation: -1`, the baseline convention) before generation 0
|
|
1068
|
-
* proposes — so even a single-generation run proposes with trace context —
|
|
1069
|
-
* and then after each generation's candidates are scored with that
|
|
1070
|
-
* generation's results; whatever it returns REPLACES `ctx.findings` for the
|
|
1071
|
-
* NEXT `propose()`, so the diagnosis is refreshed each round instead
|
|
1072
|
-
* of being a static one-shot. Generic by design: the substrate does not
|
|
1073
|
-
* import an analyst — the consumer plugs its trace-analyst registry / HALO
|
|
1074
|
-
* here (reading the per-candidate `runDir` traces). When absent, findings
|
|
1075
|
-
* stay the static `opts.findings`. */
|
|
1076
|
-
analyzeGeneration?: (input: {
|
|
1077
|
-
generation: number;
|
|
1078
|
-
runDir: string;
|
|
1079
|
-
candidates: Array<{
|
|
1080
|
-
surfaceHash: string;
|
|
1081
|
-
campaign: CampaignResult<TArtifact, TScenario>;
|
|
1082
|
-
composite: number | null;
|
|
1083
|
-
}>;
|
|
1084
|
-
history: GenerationRecord[];
|
|
1085
|
-
/** Shared run spend account and receipt attribution phase. */
|
|
1086
|
-
costLedger?: CostLedgerHandle;
|
|
1087
|
-
costPhase?: string;
|
|
1088
|
-
}) => Promise<ReadonlyArray<ProposalFinding>>;
|
|
1089
|
-
/**
|
|
1090
|
-
* Optional override for how the WINNER is selected among coverage-complete
|
|
1091
|
-
* candidates (and how the incumbent bar is set). Returns a lexicographic rank
|
|
1092
|
-
* key — each element higher-is-better; candidates are ranked by descending key
|
|
1093
|
-
* (`compareRankKeys`) and the top must STRICTLY beat the incumbent's key to
|
|
1094
|
-
* promote. Defaults to `[campaignMeanComposite(campaign)]`, i.e. the historical
|
|
1095
|
-
* scalar-mean ranking (single-element key ⇒ identical behavior).
|
|
1096
|
-
*
|
|
1097
|
-
* A binary-with-replicates consumer (e.g. swe-arena, whose ship-gate counts an
|
|
1098
|
-
* instance resolved only when EVERY replicate resolved) passes a fail-closed
|
|
1099
|
-
* key built from the SAME reduction its gate uses, so winner-selection and the
|
|
1100
|
-
* ship-gate rank on the identical metric and can never invert — the selector
|
|
1101
|
-
* cannot promote a flaky per-cell-mean candidate the gate would reject over a
|
|
1102
|
-
* fail-closed candidate the gate would accept. Only the winner CHOICE changes;
|
|
1103
|
-
* the descriptive `composite` (mean) on every record and the Pareto objective
|
|
1104
|
-
* vectors are untouched, so proposer diversity and reporting are unaffected.
|
|
1105
|
-
*/
|
|
1106
|
-
selectionRankKey?: (campaign: CampaignResult<TArtifact, TScenario>) => number[];
|
|
1107
|
-
/**
|
|
1108
|
-
* Optional policy for which scored surface the next generation MUTATES.
|
|
1109
|
-
* Absent, every generation mutates the global incumbent, so the recorded
|
|
1110
|
-
* `parentSurfaceHash` lineage is a chain. Present, the selector receives the
|
|
1111
|
-
* Pareto frontier so far, the measured incumbent, the generation history,
|
|
1112
|
-
* and the generation index, and returns one frontier parent; the loop hands
|
|
1113
|
-
* that parent to `propose()` as `currentSurface` + `parentOutcome` and
|
|
1114
|
-
* records it as every candidate's `parentSurfaceHash`. Promotion is
|
|
1115
|
-
* unchanged: a candidate still has to beat the incumbent. The loop refuses
|
|
1116
|
-
* a parent it has not measured to completion. `crowdedFrontierParent` is
|
|
1117
|
-
* the provided seeded policy.
|
|
1118
|
-
*/
|
|
1119
|
-
selectParent?: ParentSelector;
|
|
1120
|
-
/**
|
|
1121
|
-
* Record this search into a durable `SearchLedger`. The loop emits the plan,
|
|
1122
|
-
* each candidate-generation operation, each candidate registration with its
|
|
1123
|
-
* measured parent, one task attempt per designed cell, one decision per
|
|
1124
|
-
* candidate, and the terminal event, then returns a bounded
|
|
1125
|
-
* `searchHistory` receipt over the exact ledger bytes.
|
|
1126
|
-
*
|
|
1127
|
-
* `identity` declares what the ledger requires and a campaign cannot infer:
|
|
1128
|
-
* immutable revisions for the agent, proposer, and search implementations,
|
|
1129
|
-
* and the model the agent runs when a cell reports none.
|
|
1130
|
-
*/
|
|
1131
|
-
searchLedger?: SearchLedgerBinding;
|
|
1132
|
-
}
|
|
1133
|
-
type RunOptimizationOptions<TScenario extends Scenario, TArtifact> = RunOptimizationBaseOptions<TScenario, TArtifact>;
|
|
1134
|
-
interface RunOptimizationResult<TArtifact, TScenario extends Scenario> {
|
|
1135
|
-
generations: Array<{
|
|
1136
|
-
record: GenerationRecord;
|
|
1137
|
-
surfaces: Array<{
|
|
1138
|
-
surfaceHash: string;
|
|
1139
|
-
surface: MutableSurface;
|
|
1140
|
-
campaign: CampaignResult<TArtifact, TScenario>;
|
|
1141
|
-
}>;
|
|
1142
|
-
}>;
|
|
1143
|
-
/** Frozen snapshot of the exact starting surface measured by `baselineCampaign`. */
|
|
1144
|
-
baselineSurface: MutableSurface;
|
|
1145
|
-
winnerSurface: MutableSurface;
|
|
1146
|
-
winnerSurfaceHash: string;
|
|
1147
|
-
/** Proposer label for the promoted surface. Present when the winning
|
|
1148
|
-
* candidate came from a `ProposedCandidate` (a reflective proposer);
|
|
1149
|
-
* absent when the winner is the baseline or a bare-surface mutator. */
|
|
1150
|
-
winnerLabel?: string;
|
|
1151
|
-
/** Proposer rationale for the promoted surface — the "because Z" that
|
|
1152
|
-
* motivated the winning change. Survives to `SelfImproveResult` and the
|
|
1153
|
-
* emitted provenance record. Absent when the winner is the baseline. */
|
|
1154
|
-
winnerRationale?: string;
|
|
1155
|
-
baselineCampaign: CampaignResult<TArtifact, TScenario>;
|
|
1156
|
-
/** Run-wide spend, including agents, proposers, analysts, and judges. */
|
|
1157
|
-
cost: CostLedgerSummary;
|
|
1158
|
-
/** Bounded proof envelope over the canonical search ledger. Present only
|
|
1159
|
-
* when `searchLedger` was supplied. `complete` is false when the search was
|
|
1160
|
-
* interrupted or a candidate left a designed cell unscored. */
|
|
1161
|
-
searchHistory?: SearchHistoryReceipt;
|
|
1162
|
-
/** The GEPA Pareto frontier across every scored surface (baseline + all
|
|
1163
|
-
* generations) by per-scenario objective vector — the non-dominated set.
|
|
1164
|
-
* Each generation's `propose()` received the frontier-so-far as
|
|
1165
|
-
* `ctx.paretoParents`; this is the final frontier. A surface here that is
|
|
1166
|
-
* NOT the winner is uniquely best on some scenario the winner loses on. */
|
|
1167
|
-
paretoFrontier: ParetoParent[];
|
|
1168
|
-
}
|
|
1169
|
-
/**
|
|
1170
|
-
* Improvement loop body: N generations of propose → campaign → rank, maintaining a Pareto frontier and one global incumbent across generations. The parent each generation mutates is the incumbent unless `selectParent` draws it from the frontier.
|
|
1171
|
-
*/
|
|
1172
|
-
declare function runOptimization<TScenario extends Scenario, TArtifact>(opts: RunOptimizationOptions<TScenario, TArtifact>): Promise<RunOptimizationResult<TArtifact, TScenario>>;
|
|
1173
|
-
//#endregion
|
|
1174
|
-
//#region src/campaign/presets/run-improvement-loop.d.ts
|
|
1175
|
-
type RunImprovementLoopOptions<TScenario extends Scenario, TArtifact> = RunOptimizationOptions<TScenario, TArtifact> & {
|
|
1176
|
-
/** Holdout scenarios kept OUT of the training optimization pool — used
|
|
1177
|
-
* ONLY to score baseline vs winner for the gate. */
|
|
1178
|
-
holdoutScenarios: TScenario[];
|
|
1179
|
-
/** Holdout policy. Default `'measured'`: baseline + winner are re-scored on
|
|
1180
|
-
* `holdoutScenarios` and the gate decides on that held-out comparison.
|
|
1181
|
-
* `'deferred'`: the improvement-set (search) campaigns run exactly as usual,
|
|
1182
|
-
* but ZERO holdout cells are dispatched, the gate is forced to `'hold'`, and
|
|
1183
|
-
* the result + provenance record carry `holdout: 'deferred'` with NO
|
|
1184
|
-
* held-out lift — for callers that measure the held-out comparison in a
|
|
1185
|
-
* separate later run instead of faking a static holdout scenario and
|
|
1186
|
-
* recording a meaningless lift. */
|
|
1187
|
-
holdout?: 'measured' | 'deferred';
|
|
1188
|
-
/** Promotion gate. Substrate strongly recommends `defaultProductionGate`
|
|
1189
|
-
* for production wiring (composes red-team / reward-hacking / canary /
|
|
1190
|
-
* heldout). */
|
|
1191
|
-
gate: Gate<TArtifact, TScenario>;
|
|
1192
|
-
/** What to do when the gate ships:
|
|
1193
|
-
* - `'pr'`: open a PR via `openAutoPr`
|
|
1194
|
-
* - `'none'`: just report — caller decides what to do with the winner
|
|
1195
|
-
* Live-runtime self-mutation is intentionally unsupported. */
|
|
1196
|
-
autoOnPromote: 'pr' | 'none';
|
|
1197
|
-
/** GH owner / repo for the auto-PR. Required when autoOnPromote === 'pr'. */
|
|
1198
|
-
ghOwner?: string;
|
|
1199
|
-
ghRepo?: string;
|
|
1200
|
-
/** Placebo control. When supplied AND the winner differs from baseline, the
|
|
1201
|
-
* loop scores a THIRD holdout arm: the winner surface with its content
|
|
1202
|
-
* footprint-matched-blanked by this function (typically via `neutralizeText`).
|
|
1203
|
-
* Its scores are exposed to the gate as `ctx.neutralizedJudgeScores`, letting
|
|
1204
|
-
* a `neutralizationGate` reject a win whose lift survives blanking the content
|
|
1205
|
-
* (decorative — driven by footprint, not content). Costs one extra holdout
|
|
1206
|
-
* campaign; omit to skip. Return a byte/layout-matched blank of the winner. */
|
|
1207
|
-
neutralize?: (winnerSurface: MutableSurface, baselineSurface: MutableSurface) => MutableSurface;
|
|
1208
|
-
};
|
|
1209
|
-
interface RunImprovementLoopResult<TArtifact, TScenario extends Scenario> extends RunOptimizationResult<TArtifact, TScenario> {
|
|
1210
|
-
baselineOnHoldout: CampaignResult<TArtifact, TScenario>;
|
|
1211
|
-
winnerOnHoldout: CampaignResult<TArtifact, TScenario>;
|
|
1212
|
-
neutralizedOnHoldout?: CampaignResult<TArtifact, TScenario>;
|
|
1213
|
-
neutralizedSurface?: MutableSurface;
|
|
1214
|
-
gateResult: Awaited<ReturnType<Gate<TArtifact, TScenario>['decide']>>;
|
|
1215
|
-
/** Present iff the loop ran with `holdout: 'deferred'`. When set,
|
|
1216
|
-
* `baselineOnHoldout`/`winnerOnHoldout` are the shared EMPTY campaign (zero
|
|
1217
|
-
* cells dispatched) and the gate verdict is the forced `'hold'`. */
|
|
1218
|
-
holdout?: 'deferred';
|
|
1219
|
-
/** Unified baseline→winner surface diff. Computed UNCONDITIONALLY (not only
|
|
1220
|
-
* when `autoOnPromote === 'pr'`) so the diff that the gate decided on is
|
|
1221
|
-
* always present on the result + in the emitted provenance record. Empty
|
|
1222
|
-
* string when winner == baseline (no change to diff). */
|
|
1223
|
-
promotedDiff: string;
|
|
1224
|
-
prResult?: ReturnType<typeof openAutoPr>;
|
|
1225
|
-
}
|
|
1226
|
-
/**
|
|
1227
|
-
* Gated-promotion shell over `runOptimization`: scores the winner against the baseline on a holdout set, runs the release gate, and optionally opens a PR.
|
|
1228
|
-
*/
|
|
1229
|
-
declare function runImprovementLoop<TScenario extends Scenario, TArtifact>(opts: RunImprovementLoopOptions<TScenario, TArtifact>): Promise<RunImprovementLoopResult<TArtifact, TScenario>>;
|
|
1230
|
-
//#endregion
|
|
1231
|
-
//#region src/campaign/transient-failure.d.ts
|
|
1232
|
-
interface TransientFailureOptions {
|
|
1233
|
-
/**
|
|
1234
|
-
* Treat full-duration timeouts ("timeout after 180000ms") as transient.
|
|
1235
|
-
* Enable on saturated shared infrastructure where queue starvation eats
|
|
1236
|
-
* the clock; leave off when the agent had the resources and simply failed.
|
|
1237
|
-
* Default false.
|
|
1238
|
-
*/
|
|
1239
|
-
readonly retryFullDurationTimeouts?: boolean;
|
|
1240
|
-
/** Additional caller-specific transient patterns. */
|
|
1241
|
-
readonly extraPatterns?: readonly RegExp[];
|
|
1242
|
-
/**
|
|
1243
|
-
* The instant a dated quota refusal is measured against. Defaults to `Date.now()`; inject it
|
|
1244
|
-
* to replay a past classification, which is what a retry audit needs.
|
|
1245
|
-
*/
|
|
1246
|
-
readonly now?: number;
|
|
1247
|
-
}
|
|
1248
|
-
/**
|
|
1249
|
-
* The instant a provider says a spent quota works again, or null when the text states none.
|
|
1250
|
-
*
|
|
1251
|
-
* MEASURED (2026-09-01, discovery lab). The codex/ChatGPT backend answered
|
|
1252
|
-
* `You've hit your usage limit. Visit https://chatgpt.com/codex/settings/usage to purchase more
|
|
1253
|
-
* credits or try again at Sep 6th, 2026 8:29 PM.` — a refusal SIX DAYS out. 25 supervised runs met
|
|
1254
|
-
* it. 21 of them retried it 12 times over about 31 minutes and settled with zero children, zero
|
|
1255
|
-
* tokens and zero claims: about 11 hours of one subscription's capacity spent on a wall that had
|
|
1256
|
-
* already told the caller when it would come down.
|
|
1257
|
-
*
|
|
1258
|
-
* The distinction this draws is not "quota" versus "not quota". It is "the provider named a
|
|
1259
|
-
* release time" versus "it did not". A bare 429, or z.ai's `您的账户已达到速率限制`, recovers on
|
|
1260
|
-
* its own in seconds and SHOULD be retried; those return null here and keep their existing
|
|
1261
|
-
* treatment. Only a stated release is terminal, and only until that instant.
|
|
1262
|
-
*
|
|
1263
|
-
* A date with no zone is read in the host's zone, because a CLI renders it in the host's zone.
|
|
1264
|
-
* An unparseable date returns null rather than a guess: a caller that stops dispatching must
|
|
1265
|
-
* never do so on a misread string.
|
|
1266
|
-
*
|
|
1267
|
-
* @param message the provider's error text
|
|
1268
|
-
* @returns the stated release time, or null when the text names none
|
|
1269
|
-
*/
|
|
1270
|
-
declare function quotaExhaustedUntil(message: string | null | undefined): Date | null;
|
|
1271
|
-
/**
|
|
1272
|
-
* True when the error text describes an infrastructure hiccup that should be
|
|
1273
|
-
* retried rather than scored. Empty/undefined input is not transient.
|
|
1274
|
-
*/
|
|
1275
|
-
declare function isTransientTransportFailure(message: string | null | undefined, opts?: TransientFailureOptions): boolean;
|
|
1276
|
-
/**
|
|
1277
|
-
* Ready-made `cellRetry.retryable` predicate: true for a dispatch-stage
|
|
1278
|
-
* failure whose error message `isTransientTransportFailure` classifies as an
|
|
1279
|
-
* infrastructure hiccup. A judge-stage failure is never retried here — the
|
|
1280
|
-
* dispatch already produced an artifact, so re-dispatching would score a
|
|
1281
|
-
* different sample. A per-cell dispatch deadline ("dispatch exceeded <N>ms")
|
|
1282
|
-
* is not transient by default; opt in via `extraPatterns` or
|
|
1283
|
-
* `retryFullDurationTimeouts` when queue starvation eats the clock. A provider refusal that
|
|
1284
|
-
* states its own release time is never retried while that time is in the future
|
|
1285
|
-
* (`quotaExhaustedUntil`).
|
|
1286
|
-
*/
|
|
1287
|
-
declare function transientDispatchFailure(opts?: TransientFailureOptions): (failure: CampaignCellFailureReceipt['failure']) => boolean;
|
|
1288
|
-
//#endregion
|
|
1289
|
-
//#region src/llm-judge.d.ts
|
|
1290
|
-
/** A rubric dimension as a bare key or the full `{ key, description }` shape. A
|
|
1291
|
-
* bare string uses the key as its own description. */
|
|
1292
|
-
type LlmJudgeDimension = string | JudgeDimension;
|
|
1293
|
-
interface LlmJudgeOptions<TArtifact, TScenario extends Scenario = Scenario> {
|
|
1294
|
-
/** The injected LLM transport. One `chat()` call per `score()`. Required —
|
|
1295
|
-
* there is no default route, so a misconfigured judge fails at construction,
|
|
1296
|
-
* never silently against the free-tier router. */
|
|
1297
|
-
chat: ChatClient;
|
|
1298
|
-
/** Rubric dimensions the model scores. Each becomes a `[0,1]` field of the
|
|
1299
|
-
* returned `JudgeScore.dimensions`. Defaults to a single `quality` dimension. */
|
|
1300
|
-
dimensions?: LlmJudgeDimension[];
|
|
1301
|
-
/** Model id. Falls back to `chat.defaultModel`; one of the two MUST resolve. */
|
|
1302
|
-
model?: string;
|
|
1303
|
-
/** Explicit scoring revision for opaque transport or renderer changes. */
|
|
1304
|
-
judgeVersion?: string;
|
|
1305
|
-
temperature?: number;
|
|
1306
|
-
maxTokens?: number;
|
|
1307
|
-
/** Composite weights forwarded to `weightedComposite`: a partial map selects
|
|
1308
|
-
* AND weights exactly the named dimensions. Omit for a uniform mean. */
|
|
1309
|
-
weights?: Record<string, number>;
|
|
1310
|
-
/**
|
|
1311
|
-
* How to read a score out of the model's answer.
|
|
1312
|
-
*
|
|
1313
|
-
* `'sampled'` (default) reads the number the model emitted. Discrete grades
|
|
1314
|
-
* tie often, and a tie carries no ranking signal.
|
|
1315
|
-
*
|
|
1316
|
-
* `'expectation'` asks the provider for the log probabilities of the score
|
|
1317
|
-
* token and returns the expected value over the integer grades the model
|
|
1318
|
-
* considered, so two answers that both sample `8` separate by how much mass
|
|
1319
|
-
* sat on `7` and `9`. It requires `scale: 'ten'`: an integer grade is one
|
|
1320
|
-
* token, and a `unit` float is not. `whenUnavailable` decides what happens
|
|
1321
|
-
* when the provider returns no log probabilities, or the grade did not land
|
|
1322
|
-
* in one token: `'fail'` throws, `'sampled'` reads the emitted number and
|
|
1323
|
-
* records `scoringMethod: 'sampled'` on the score.
|
|
1324
|
-
*/
|
|
1325
|
-
scoring?: {
|
|
1326
|
-
method: 'sampled';
|
|
1327
|
-
} | {
|
|
1328
|
-
method: 'expectation';
|
|
1329
|
-
whenUnavailable: 'fail' | 'sampled';
|
|
1330
|
-
};
|
|
1331
|
-
/** Scale the model is prompted to score on, normalized into `[0,1]`:
|
|
1332
|
-
* - `'unit'` (default): the model returns `[0,1]` directly.
|
|
1333
|
-
* - `'ten'`: the model returns `[0,10]`; divided by 10 here.
|
|
1334
|
-
* The prompt is annotated with the expected range either way. */
|
|
1335
|
-
scale?: 'unit' | 'ten';
|
|
1336
|
-
/** Run this judge only on matching scenarios (mirrors `JudgeConfig.appliesTo`). */
|
|
1337
|
-
appliesTo?: (scenario: TScenario) => boolean;
|
|
1338
|
-
/** Render the artifact + scenario into the user message. Default:
|
|
1339
|
-
* pretty-printed JSON of `{ scenario, artifact }`. */
|
|
1340
|
-
renderUser?: (input: {
|
|
1341
|
-
artifact: TArtifact;
|
|
1342
|
-
scenario: TScenario;
|
|
1343
|
-
}) => string;
|
|
1344
|
-
/** Strict runtime contract; its JSON Schema is sent to the provider. */
|
|
1345
|
-
costLedger?: CostLedgerHandle;
|
|
1346
|
-
responseSchema?: {
|
|
1347
|
-
name: string;
|
|
1348
|
-
schema: z.ZodObject;
|
|
1349
|
-
};
|
|
1350
|
-
}
|
|
1351
|
-
/**
|
|
1352
|
-
* Build a campaign-shaped `JudgeConfig` whose `score()` makes ONE LLM call
|
|
1353
|
-
* against `prompt` and reduces the model's per-dimension scores to a canonical
|
|
1354
|
-
* `JudgeScore` in `[0,1]`.
|
|
1355
|
-
*
|
|
1356
|
-
* The model is instructed to return JSON `{ "dimensions": { <key>: <number>, … },
|
|
1357
|
-
* "notes": "…" }`; the helper strips fenced JSON, validates every declared
|
|
1358
|
-
* dimension is present and in range, normalizes by `scale`, and composites via
|
|
1359
|
-
* `weightedComposite`.
|
|
1360
|
-
*/
|
|
1361
|
-
declare function llmJudge<TArtifact = unknown, TScenario extends Scenario = Scenario>(name: string, prompt: string, opts: LlmJudgeOptions<TArtifact, TScenario>): JudgeConfig<TArtifact, TScenario>;
|
|
1362
|
-
//#endregion
|
|
1363
|
-
//#region src/campaign/external-optimizer-observations.d.ts
|
|
1364
|
-
interface ExternalOptimizerObservationSummary {
|
|
1365
|
-
scope: 'callback-submitted-candidates';
|
|
1366
|
-
path: string;
|
|
1367
|
-
sha256: `sha256:${string}`;
|
|
1368
|
-
submittedCandidates: number;
|
|
1369
|
-
evaluations: number;
|
|
1370
|
-
refusals: number;
|
|
1371
|
-
}
|
|
1372
|
-
interface ExternalOptimizerExecutionSummary {
|
|
1373
|
-
scope: 'runtime-model-calls';
|
|
1374
|
-
path: string;
|
|
1375
|
-
sha256: `sha256:${string}`;
|
|
1376
|
-
calls: number;
|
|
1377
|
-
succeeded: number;
|
|
1378
|
-
failed: number;
|
|
1379
|
-
}
|
|
1380
|
-
interface ExternalOptimizerSubmittedCandidate {
|
|
1381
|
-
/** Exact text or named-component surface submitted to the evaluation callback. */
|
|
1382
|
-
readonly candidate: ExternalTextCandidate;
|
|
1383
|
-
/** Eval's canonical content identity for `candidate`. */
|
|
1384
|
-
readonly candidateHash: string;
|
|
1385
|
-
readonly candidateDigest: `sha256:${string}`;
|
|
1386
|
-
readonly proposalSequence: number;
|
|
1387
|
-
/** Exact observation artifact that proves this candidate was submitted. */
|
|
1388
|
-
readonly provenance: {
|
|
1389
|
-
readonly path: string;
|
|
1390
|
-
readonly sha256: `sha256:${string}`;
|
|
1391
|
-
};
|
|
1392
|
-
}
|
|
1393
|
-
interface ExternalOptimizerObservationArtifact {
|
|
1394
|
-
readonly summary: ExternalOptimizerObservationSummary;
|
|
1395
|
-
readonly observations: readonly ExternalOptimizerEvaluationObservation[];
|
|
1396
|
-
/** Every distinct callback-submitted candidate in proposal order. */
|
|
1397
|
-
readonly candidates: readonly ExternalOptimizerSubmittedCandidate[];
|
|
1398
|
-
}
|
|
1399
|
-
/**
|
|
1400
|
-
* Read and verify the exact callback observation artifact addressed by method provenance.
|
|
1401
|
-
*
|
|
1402
|
-
* The reader checks the raw SHA-256, canonical JSONL bytes, sequence, candidate
|
|
1403
|
-
* identities, and summary counts before it returns any candidate.
|
|
1404
|
-
* This proves that the bytes match the supplied summary. The caller remains
|
|
1405
|
-
* responsible for obtaining that summary from trusted provenance.
|
|
1406
|
-
*/
|
|
1407
|
-
declare function readExternalOptimizerObservationArtifact(input: {
|
|
1408
|
-
summary: ExternalOptimizerObservationSummary;
|
|
1409
|
-
storage?: CampaignStorage;
|
|
1410
|
-
}): ExternalOptimizerObservationArtifact;
|
|
1411
|
-
//#endregion
|
|
1412
|
-
//#region src/campaign/presets/compare-optimization-methods.d.ts
|
|
1413
|
-
/** Shared campaign settings applied to every optimization method. */
|
|
1414
|
-
type OptimizationMethodRunOptions<TScenario extends Scenario, TArtifact> = Omit<RunCampaignOptions<TScenario, TArtifact>, 'costCeiling' | 'costLedger' | 'dispatch' | 'judges' | 'runDir' | 'scenarios' | 'seed'>;
|
|
1415
|
-
/** Cost reported by a method or by final test scoring. */
|
|
1416
|
-
interface ComparisonCost {
|
|
1417
|
-
/** Known subtotal. Consult `costProvenance` before treating this as total spend. */
|
|
1418
|
-
totalCostUsd: number;
|
|
1419
|
-
/** Exact origin of the total; uncaptured means `totalCostUsd` is only a known subtotal. */
|
|
1420
|
-
costProvenance: CostProvenance;
|
|
1421
|
-
accountingComplete: boolean;
|
|
1422
|
-
incompleteReasons: string[];
|
|
1423
|
-
}
|
|
1424
|
-
interface OptimizationPackageSource {
|
|
1425
|
-
kind: 'package';
|
|
1426
|
-
/** Whether package identity was inspected or supplied by caller code. */
|
|
1427
|
-
evidence: 'observed' | 'declared';
|
|
1428
|
-
package: string;
|
|
1429
|
-
version: string;
|
|
1430
|
-
sourceUrl?: string;
|
|
1431
|
-
revision?: string;
|
|
1432
|
-
/** SHA-256 of all installed module files observed before the run. */
|
|
1433
|
-
sourceSha256?: string;
|
|
1434
|
-
}
|
|
1435
|
-
interface OptimizationModuleSource {
|
|
1436
|
-
module: string;
|
|
1437
|
-
sourceSha256: string;
|
|
1438
|
-
}
|
|
1439
|
-
interface OptimizationPythonRuntime {
|
|
1440
|
-
implementation: string;
|
|
1441
|
-
version: string;
|
|
1442
|
-
}
|
|
1443
|
-
interface OptimizationTokenUsage {
|
|
1444
|
-
/** All input tokens, including cache reads and cache creation. */
|
|
1445
|
-
inputTokens: number;
|
|
1446
|
-
/** Input tokens served from a provider cache. */
|
|
1447
|
-
cachedInputTokens?: number;
|
|
1448
|
-
/** Input tokens used to create or write a provider cache entry. */
|
|
1449
|
-
cacheWriteInputTokens?: number;
|
|
1450
|
-
outputTokens: number;
|
|
1451
|
-
/** Reasoning tokens included in `outputTokens`. */
|
|
1452
|
-
reasoningTokens?: number;
|
|
1453
|
-
totalTokens: number;
|
|
1454
|
-
calls: number;
|
|
1455
|
-
}
|
|
1456
|
-
interface OptimizationMethodProvenance {
|
|
1457
|
-
/** External optimizer package. */
|
|
1458
|
-
source: OptimizationPackageSource;
|
|
1459
|
-
/** Python bridge package that invoked the optimizer. */
|
|
1460
|
-
bridge?: OptimizationPackageSource;
|
|
1461
|
-
/** Custom engine modules imported by the optimizer. */
|
|
1462
|
-
modules?: OptimizationModuleSource[];
|
|
1463
|
-
/** Python implementation used by the bridge process. */
|
|
1464
|
-
python?: OptimizationPythonRuntime;
|
|
1465
|
-
/** Exact model identifier configured for optimizer-owned model calls. */
|
|
1466
|
-
optimizerModel?: string;
|
|
1467
|
-
/** Stable public identity of the execution-owner callback. */
|
|
1468
|
-
optimizerCallRef?: string;
|
|
1469
|
-
runId: string;
|
|
1470
|
-
/** Content identity shared by compatible resumptions. */
|
|
1471
|
-
compatibleRunId?: string;
|
|
1472
|
-
resumed: boolean;
|
|
1473
|
-
/** Whether the run seed reached every external engine configuration. */
|
|
1474
|
-
seedApplied?: boolean;
|
|
1475
|
-
/** Evaluations the local callback metered — the trusted count. */
|
|
1476
|
-
evaluationCount: number;
|
|
1477
|
-
/**
|
|
1478
|
-
* Evaluation total the external optimizer reported from its own counters.
|
|
1479
|
-
* A difference from `evaluationCount` means upstream skipped, cached, or
|
|
1480
|
-
* double-counted work; inspect before trusting upstream-derived budgets.
|
|
1481
|
-
*/
|
|
1482
|
-
upstreamReportedEvaluations?: number;
|
|
1483
|
-
artifactDir: string;
|
|
1484
|
-
tokenUsage?: OptimizationTokenUsage;
|
|
1485
|
-
/** Candidates submitted to the callback, per-case scores, and refusals. */
|
|
1486
|
-
observations?: ExternalOptimizerObservationSummary;
|
|
1487
|
-
/** Exact accepted GEPA candidates, parent indices, and selection scores. */
|
|
1488
|
-
gepaCandidatePopulation?: GepaCandidatePopulationSummary;
|
|
1489
|
-
/** Opaque Runtime execution evidence for every invoked optimizer-model call. */
|
|
1490
|
-
modelExecutions?: ExternalOptimizerExecutionSummary;
|
|
1491
|
-
/** Anthropic-endpoint proxy traffic from agent CLI engines, when enabled. */
|
|
1492
|
-
anthropicEndpoint?: ExternalOptimizerWireCounts;
|
|
1493
|
-
}
|
|
1494
|
-
/** Shared inputs for one optimization method. Final test data is absent. */
|
|
1495
|
-
interface OptimizationMethodInput<TScenario extends Scenario, TArtifact> {
|
|
1496
|
-
/** Surface every method starts from. */
|
|
1497
|
-
readonly baselineSurface: MutableSurface;
|
|
1498
|
-
/** Evidence used to author or fit candidates. */
|
|
1499
|
-
readonly trainScenarios: readonly TScenario[];
|
|
1500
|
-
/** Data used for candidate acceptance, early stopping, and model selection. */
|
|
1501
|
-
readonly selectionScenarios: readonly TScenario[];
|
|
1502
|
-
/** Runs one scenario with a candidate surface. */
|
|
1503
|
-
readonly dispatchWithSurface: (surface: MutableSurface, scenario: TScenario, ctx: DispatchContext) => Promise<TArtifact>;
|
|
1504
|
-
/** Scores artifacts produced by `dispatchWithSurface`. */
|
|
1505
|
-
readonly judges: readonly JudgeConfig<TArtifact, TScenario>[];
|
|
1506
|
-
/** Method-specific artifacts are written below this directory. */
|
|
1507
|
-
readonly runDir: string;
|
|
1508
|
-
readonly seed: number;
|
|
1509
|
-
/** Shared defaults for every method. A method may override them explicitly. */
|
|
1510
|
-
readonly runOptions: Readonly<OptimizationMethodRunOptions<TScenario, TArtifact>>;
|
|
1511
|
-
/** Durable spend account shared by every method and final scoring. */
|
|
1512
|
-
readonly costLedger: CostLedgerHandle;
|
|
1513
|
-
}
|
|
1514
|
-
interface OptimizationMethodResult {
|
|
1515
|
-
/** Surface selected without using the final test partition. */
|
|
1516
|
-
winnerSurface: MutableSurface;
|
|
1517
|
-
/** Optimization spend. Excludes final test scoring. */
|
|
1518
|
-
cost: ComparisonCost;
|
|
1519
|
-
/** Optimization duration. Excludes final test scoring. */
|
|
1520
|
-
durationMs?: number;
|
|
1521
|
-
/** Exact external implementation and run identity, when the method uses one. */
|
|
1522
|
-
provenance?: OptimizationMethodProvenance;
|
|
1523
|
-
/** Bounded proof envelope over the canonical SearchLedger for this optimization. */
|
|
1524
|
-
searchHistory?: SearchHistoryReceipt;
|
|
1525
|
-
}
|
|
1526
|
-
/** A complete optimization method, including candidate generation and selection. */
|
|
1527
|
-
interface OptimizationMethod<TScenario extends Scenario = Scenario, TArtifact = unknown> {
|
|
1528
|
-
/** Unique, trimmed display name. Its normalized form must also be unique. */
|
|
1529
|
-
name: string;
|
|
1530
|
-
optimize: (input: OptimizationMethodInput<TScenario, TArtifact>) => Promise<OptimizationMethodResult>;
|
|
1531
|
-
}
|
|
1532
|
-
interface OptimizationMethodScore {
|
|
1533
|
-
name: string;
|
|
1534
|
-
/** Mean final-test composite of the baseline (identical across methods). */
|
|
1535
|
-
baselineComposite: number;
|
|
1536
|
-
/** Mean final-test composite of this method's selected surface. */
|
|
1537
|
-
winnerComposite: number;
|
|
1538
|
-
/** Mean per-scenario final-test lift (winner minus baseline). */
|
|
1539
|
-
lift: number;
|
|
1540
|
-
/** Simultaneous paired-bootstrap interval for per-scenario lift.
|
|
1541
|
-
* `low > 0` excludes zero after adjustment for all reported contrasts. */
|
|
1542
|
-
liftCi: {
|
|
1543
|
-
low: number;
|
|
1544
|
-
high: number;
|
|
1545
|
-
};
|
|
1546
|
-
/** Optimization spend reported by the method. Excludes final test scoring. */
|
|
1547
|
-
optimizationCost: ComparisonCost;
|
|
1548
|
-
/** Optimization duration reported by the method. Excludes final test scoring. */
|
|
1549
|
-
durationMs?: number;
|
|
1550
|
-
/** Exact external implementation and run identity, when reported by the method. */
|
|
1551
|
-
provenance?: OptimizationMethodProvenance;
|
|
1552
|
-
/** Paired final-test values used to compute lift and its interval. */
|
|
1553
|
-
scenarioScores: Array<{
|
|
1554
|
-
scenarioId: string;
|
|
1555
|
-
baselineComposite: number;
|
|
1556
|
-
winnerComposite: number;
|
|
1557
|
-
lift: number;
|
|
1558
|
-
}>;
|
|
1559
|
-
winnerSurface: MutableSurface;
|
|
1560
|
-
/** 1-based, by descending lift. */
|
|
1561
|
-
rank: number;
|
|
1562
|
-
}
|
|
1563
|
-
interface OptimizationMethodPairwise {
|
|
1564
|
-
/** Higher-ranked method. */
|
|
1565
|
-
a: string;
|
|
1566
|
-
b: string;
|
|
1567
|
-
/** Mean per-scenario untouched-test delta (a − b). */
|
|
1568
|
-
deltaMean: number;
|
|
1569
|
-
low: number;
|
|
1570
|
-
high: number;
|
|
1571
|
-
/** `a` if the CI clears 0, `b` if it is entirely negative, else `'tie'`. */
|
|
1572
|
-
favored: string;
|
|
1573
|
-
}
|
|
1574
|
-
interface OptimizationMethodComparison {
|
|
1575
|
-
/** Sorted by descending lift; `rank` set accordingly. */
|
|
1576
|
-
scores: OptimizationMethodScore[];
|
|
1577
|
-
best: OptimizationMethodScore;
|
|
1578
|
-
/** Best vs each other method, using simultaneous paired-bootstrap intervals. */
|
|
1579
|
-
pairwise: OptimizationMethodPairwise[];
|
|
1580
|
-
testScenarioIds: string[];
|
|
1581
|
-
/** Sum of the costs reported by every optimization method. */
|
|
1582
|
-
optimizationCost: ComparisonCost;
|
|
1583
|
-
/** Baseline and distinct winner scoring on the final test partition. */
|
|
1584
|
-
testCost: ComparisonCost;
|
|
1585
|
-
/** Optimization plus final test scoring. */
|
|
1586
|
-
totalCost: ComparisonCost;
|
|
1587
|
-
/** Caller-requested simultaneous coverage across all reported contrasts. */
|
|
1588
|
-
confidence: number;
|
|
1589
|
-
/** Bonferroni-adjusted confidence used for each bootstrap interval. */
|
|
1590
|
-
intervalConfidence: number;
|
|
1591
|
-
/** Method-vs-baseline plus all possible method-vs-method contrasts. */
|
|
1592
|
-
comparisonCount: number;
|
|
1593
|
-
/** Deterministic bootstrap and campaign seed. */
|
|
1594
|
-
seed: number;
|
|
1595
|
-
/** Bootstrap draws used for each interval. */
|
|
1596
|
-
resamples: number;
|
|
1597
|
-
/** Agent runs averaged within each test scenario before resampling scenarios. */
|
|
1598
|
-
reps: number;
|
|
1599
|
-
/** Coverage of every method's canonical search history. */
|
|
1600
|
-
searchHistory: SearchHistoryCoverage;
|
|
1601
|
-
}
|
|
1602
|
-
interface CompareOptimizationMethodsOptions<TScenario extends Scenario, TArtifact> extends Omit<RunCampaignOptions<TScenario, TArtifact>, 'dispatch' | 'judges' | 'scenarios'> {
|
|
1603
|
-
methods: OptimizationMethod<TScenario, TArtifact>[];
|
|
1604
|
-
baselineSurface: MutableSurface;
|
|
1605
|
-
/** Evidence used by every optimizer to author or fit candidates. */
|
|
1606
|
-
trainScenarios: TScenario[];
|
|
1607
|
-
/** Candidate acceptance, early-stopping, and optimizer-selection data. */
|
|
1608
|
-
selectionScenarios: TScenario[];
|
|
1609
|
-
/** Untouched final comparison data. Never passed to an optimization method. */
|
|
1610
|
-
testScenarios: TScenario[];
|
|
1611
|
-
/** Scores a surface on a scenario. The methods and final test share this function. */
|
|
1612
|
-
dispatchWithSurface: (surface: MutableSurface, scenario: TScenario, ctx: DispatchContext) => Promise<TArtifact>;
|
|
1613
|
-
judges: JudgeConfig<TArtifact, TScenario>[];
|
|
1614
|
-
/** Bootstrap resamples for the lift intervals. Default is at least 2000 and
|
|
1615
|
-
* rises when the requested simultaneous confidence needs finer tails. */
|
|
1616
|
-
resamples?: number;
|
|
1617
|
-
/** Shared defaults for each method's train and selection campaigns. */
|
|
1618
|
-
optimizationRunOptions?: OptimizationMethodRunOptions<TScenario, TArtifact>;
|
|
1619
|
-
/** Number of optimization methods to run concurrently. Default 1. */
|
|
1620
|
-
optimizationConcurrency?: number;
|
|
1621
|
-
/** Simultaneous confidence across method-vs-baseline and method-vs-method contrasts.
|
|
1622
|
-
* Each bootstrap interval is Bonferroni-adjusted. Default 0.95. */
|
|
1623
|
-
confidence?: number;
|
|
1624
|
-
/** Shared spend limit across every method's optimizer and evaluation calls plus final scoring. */
|
|
1625
|
-
costCeiling?: number;
|
|
1626
|
-
/**
|
|
1627
|
-
* Missing history is reported by default. Publication-grade or autonomous
|
|
1628
|
-
* callers set `require-complete`, which aborts before the first final-test call.
|
|
1629
|
-
*/
|
|
1630
|
-
searchHistoryPolicy?: SearchHistoryPolicy;
|
|
1631
|
-
}
|
|
1632
|
-
/**
|
|
1633
|
-
* Compare complete optimization methods on disjoint train, selection, and final test data.
|
|
1634
|
-
*/
|
|
1635
|
-
declare function compareOptimizationMethods<TScenario extends Scenario, TArtifact>(opts: CompareOptimizationMethodsOptions<TScenario, TArtifact>): Promise<OptimizationMethodComparison>;
|
|
1636
|
-
/** Keep the cost fields a custom optimization method must report. */
|
|
1637
|
-
declare function costFromLedgerSummary(summary: CostLedgerSummary): ComparisonCost;
|
|
1638
|
-
/** Preserve every optimizer token class while keeping total input and output explicit. */
|
|
1639
|
-
declare function optimizationTokenUsageFromSummary(summary: CostLedgerSummary, receipts: readonly CostReceipt[]): OptimizationTokenUsage | undefined;
|
|
1640
|
-
/** Combine method costs without turning one unknown bill into a known total. */
|
|
1641
|
-
declare function combineComparisonCosts(entries: ReadonlyArray<{
|
|
1642
|
-
label: string;
|
|
1643
|
-
cost: ComparisonCost;
|
|
1644
|
-
}>): ComparisonCost;
|
|
1645
|
-
//#endregion
|
|
1646
|
-
//#region src/canary.d.ts
|
|
1647
|
-
type CanaryKind = 'silent_judge_fallback' | 'judge_calibration_drift' | 'distribution_shift';
|
|
1648
|
-
type CanarySeverity = 'info' | 'warn' | 'error';
|
|
1649
|
-
interface CanaryAlert {
|
|
1650
|
-
kind: CanaryKind;
|
|
1651
|
-
severity: CanarySeverity;
|
|
1652
|
-
message: string;
|
|
1653
|
-
/** Numbers that informed the decision — drop straight into a
|
|
1654
|
-
* dashboard / paper figure. */
|
|
1655
|
-
evidence: Record<string, unknown>;
|
|
1656
|
-
}
|
|
1657
|
-
interface CanaryReport {
|
|
1658
|
-
alerts: CanaryAlert[];
|
|
1659
|
-
/** Per-kind summary count. */
|
|
1660
|
-
counts: Record<CanaryKind, number>;
|
|
1661
|
-
/** Whether each enabled detector had enough observations to run. */
|
|
1662
|
-
evaluations: CanaryEvaluation[];
|
|
1663
|
-
}
|
|
1664
|
-
interface CanaryEvaluation {
|
|
1665
|
-
kind: CanaryKind;
|
|
1666
|
-
status: 'evaluated' | 'not_evaluated';
|
|
1667
|
-
observations: number;
|
|
1668
|
-
reason?: string;
|
|
1669
|
-
}
|
|
1670
|
-
interface CanaryOptions {
|
|
1671
|
-
/**
|
|
1672
|
-
* Silent-fallback detection.
|
|
1673
|
-
* - `constant`: confidence value treated as the fallback signal.
|
|
1674
|
-
* Default 0.30 (matches the soft-fail default in
|
|
1675
|
-
* `propose-review.ts`).
|
|
1676
|
-
* - `consecutiveThreshold`: trip the alert after this many
|
|
1677
|
-
* consecutive runs at `constant` (or `fallback === true`).
|
|
1678
|
-
* Default 3.
|
|
1679
|
-
*/
|
|
1680
|
-
silentFallback?: {
|
|
1681
|
-
constant?: number;
|
|
1682
|
-
consecutiveThreshold?: number;
|
|
1683
|
-
/** Floating-point tolerance when comparing against `constant`. */
|
|
1684
|
-
epsilon?: number;
|
|
1685
|
-
};
|
|
1686
|
-
/**
|
|
1687
|
-
* Calibration-drift detection.
|
|
1688
|
-
* - `historyWindow`: number of past runs (oldest-first) treated as
|
|
1689
|
-
* the historical baseline. Default 50.
|
|
1690
|
-
* - `recentWindow`: number of recent runs (newest-first) compared
|
|
1691
|
-
* against history. Default 20.
|
|
1692
|
-
* - `ksAlpha`: alpha for the KS statistic vs critical value.
|
|
1693
|
-
* Default 0.05.
|
|
1694
|
-
* - `minRecent`: minimum recent runs required to even attempt the
|
|
1695
|
-
* check. Default 10.
|
|
1696
|
-
*/
|
|
1697
|
-
calibrationDrift?: {
|
|
1698
|
-
historyWindow?: number;
|
|
1699
|
-
recentWindow?: number;
|
|
1700
|
-
ksAlpha?: number;
|
|
1701
|
-
minRecent?: number;
|
|
1702
|
-
};
|
|
1703
|
-
/**
|
|
1704
|
-
* Distribution-shift detection.
|
|
1705
|
-
* - `category`: function that maps a run to a categorical bucket.
|
|
1706
|
-
* Required to enable this canary; if omitted the chi-square check
|
|
1707
|
-
* is skipped entirely.
|
|
1708
|
-
* - `chiSquareAlpha`: alpha. Default 0.05.
|
|
1709
|
-
* - `historyWindow`, `recentWindow`, `minRecent`: like above.
|
|
1710
|
-
*/
|
|
1711
|
-
distributionShift?: {
|
|
1712
|
-
category: (run: RunRecord) => string | null;
|
|
1713
|
-
chiSquareAlpha?: number;
|
|
1714
|
-
historyWindow?: number;
|
|
1715
|
-
recentWindow?: number;
|
|
1716
|
-
minRecent?: number;
|
|
1717
|
-
};
|
|
1718
|
-
}
|
|
1719
|
-
/**
|
|
1720
|
-
* Run all configured canaries against a chronological run list.
|
|
1721
|
-
* Runs MUST be sorted oldest-to-newest by the caller — the order of
|
|
1722
|
-
* the input is used to define "recent" vs "historical" windows.
|
|
1723
|
-
*/
|
|
1724
|
-
declare function runCanaries(runs: RunRecord[], opts?: CanaryOptions): CanaryReport;
|
|
1725
|
-
//#endregion
|
|
1726
|
-
//#region src/red-team.d.ts
|
|
1727
|
-
type RedTeamCategory = 'prompt_injection_direct' | 'prompt_injection_indirect' | 'jailbreak_persona' | 'jailbreak_dan' | 'pii_leak' | 'permission_escalation' | 'data_exfil' | 'policy_override';
|
|
1728
|
-
interface RedTeamPayload {
|
|
1729
|
-
category: RedTeamCategory;
|
|
1730
|
-
/** The hostile input the agent sees. */
|
|
1731
|
-
input: string;
|
|
1732
|
-
/**
|
|
1733
|
-
* What a safe agent should do: "refuse" means explicit refusal,
|
|
1734
|
-
* "ignore" means process benignly without executing the attack.
|
|
1735
|
-
*/
|
|
1736
|
-
expected: 'refuse' | 'ignore';
|
|
1737
|
-
/** Substring(s) that MUST NOT appear in output. */
|
|
1738
|
-
forbiddenStrings?: string[];
|
|
1739
|
-
/** Tool names the agent MUST NOT call for this scenario. */
|
|
1740
|
-
forbiddenTools?: string[];
|
|
1741
|
-
}
|
|
1742
|
-
interface RedTeamCase extends DatasetScenario {
|
|
1743
|
-
payload: RedTeamPayload;
|
|
1744
|
-
}
|
|
1745
|
-
interface RedTeamFinding {
|
|
1746
|
-
scenarioId: string;
|
|
1747
|
-
category: RedTeamCategory;
|
|
1748
|
-
passed: boolean;
|
|
1749
|
-
reason: string;
|
|
1750
|
-
evidence?: string;
|
|
1751
|
-
}
|
|
1752
|
-
interface RedTeamReport {
|
|
1753
|
-
findings: RedTeamFinding[];
|
|
1754
|
-
passRateByCategory: Record<RedTeamCategory, number>;
|
|
1755
|
-
overallPassRate: number;
|
|
1756
|
-
}
|
|
1757
|
-
/** Ship a canonical, small corpus. Consumers extend via `extendCorpus`. */
|
|
1758
|
-
declare const DEFAULT_RED_TEAM_CORPUS: RedTeamCase[];
|
|
1759
|
-
declare function redTeamDataset(extraCases?: RedTeamCase[]): Dataset;
|
|
1760
|
-
/**
|
|
1761
|
-
* Score a single agent response against the red-team case.
|
|
1762
|
-
* `toolCalls` should be the names of tools the agent invoked during the run.
|
|
1763
|
-
*/
|
|
1764
|
-
declare function scoreRedTeamOutput(output: string, toolCalls: string[], rtCase: RedTeamCase): RedTeamFinding;
|
|
1765
|
-
/** Aggregate red-team findings into per-category pass rates. */
|
|
1766
|
-
declare function redTeamReport(findings: RedTeamFinding[]): RedTeamReport;
|
|
1767
|
-
//#endregion
|
|
1768
|
-
//#region src/campaign/provenance.d.ts
|
|
1769
|
-
interface LoopProvenanceCandidate {
|
|
1770
|
-
/** Generation index this candidate was proposed in. */
|
|
1771
|
-
generation: number;
|
|
1772
|
-
/** 16-char loop-identity fingerprint (matches `GenerationCandidate.surfaceHash`). */
|
|
1773
|
-
surfaceHash: string;
|
|
1774
|
-
/** Full sha256 content hash — byte-identical-verifiable. */
|
|
1775
|
-
contentHash: string;
|
|
1776
|
-
/** Exact scored rows that produced this candidate's search result. */
|
|
1777
|
-
campaignDigest: `sha256:${string}`;
|
|
1778
|
-
/** Proposer label, when the proposer returned a `ProposedCandidate`. */
|
|
1779
|
-
label?: string;
|
|
1780
|
-
/** Proposer rationale — the "because Z". When the proposer returned a bare
|
|
1781
|
-
* surface (blind mutator) this is absent. */
|
|
1782
|
-
rationale?: string;
|
|
1783
|
-
/** Proposer-supplied typed attribution, carried unchanged from
|
|
1784
|
-
* `GenerationCandidate.attribution`. Opaque here; the producer's schema tag
|
|
1785
|
-
* governs interpretation. */
|
|
1786
|
-
attribution?: Readonly<Record<string, unknown>>;
|
|
1787
|
-
/** Exact complete incumbent this candidate mutated. */
|
|
1788
|
-
parentSurfaceHash: string;
|
|
1789
|
-
/** Search-split composite of the exact parent. */
|
|
1790
|
-
parentComposite: number;
|
|
1791
|
-
/** Search-split composite change relative to the exact parent. */
|
|
1792
|
-
observedDeltaFromParent?: number;
|
|
1793
|
-
/** Whether the candidate completed every designed cell and could be selected. */
|
|
1794
|
-
eligibleForPromotion: boolean;
|
|
1795
|
-
/** Designed-denominator receipt retained even for incomplete candidates. */
|
|
1796
|
-
coverage: NonNullable<GenerationCandidate['coverage']>;
|
|
1797
|
-
/** Mean composite this candidate scored on the search split, or null when unscorable. */
|
|
1798
|
-
composite: number | null;
|
|
1799
|
-
/** Whether this candidate was promoted out of its generation. */
|
|
1800
|
-
promoted: boolean;
|
|
1801
|
-
}
|
|
1802
|
-
interface LoopProvenanceBackend {
|
|
1803
|
-
/** `assertRealBackend`-grade verdict over the worker call records. */
|
|
1804
|
-
verdict: 'real' | 'mixed' | 'stub';
|
|
1805
|
-
/** Number of worker LLM calls captured (the audit's "worker call count"). */
|
|
1806
|
-
workerCallCount: number;
|
|
1807
|
-
/** Distinct model ids observed across worker calls. */
|
|
1808
|
-
models: string[];
|
|
1809
|
-
totalInputTokens: number;
|
|
1810
|
-
totalOutputTokens: number;
|
|
1811
|
-
totalCostUsd: number;
|
|
1812
|
-
}
|
|
1813
|
-
interface LoopProvenanceEvidence {
|
|
1814
|
-
search: {
|
|
1815
|
-
splitDigest: `sha256:${string}`;
|
|
1816
|
-
baselineCampaignDigest: `sha256:${string}`;
|
|
1817
|
-
};
|
|
1818
|
-
holdout: {
|
|
1819
|
-
splitDigest: `sha256:${string}`;
|
|
1820
|
-
baselineCampaignDigest: `sha256:${string}`;
|
|
1821
|
-
winnerCampaignDigest: `sha256:${string}`;
|
|
1822
|
-
neutralized?: {
|
|
1823
|
-
contentHash: `sha256:${string}`;
|
|
1824
|
-
campaignDigest: `sha256:${string}`;
|
|
1825
|
-
composite: number;
|
|
1826
|
-
lift: number;
|
|
1827
|
-
};
|
|
1828
|
-
};
|
|
1829
|
-
costReceiptsDigest: `sha256:${string}`;
|
|
1830
|
-
}
|
|
1831
|
-
interface LoopProvenanceOptimizationMethod {
|
|
1832
|
-
name: string;
|
|
1833
|
-
cost: ComparisonCost;
|
|
1834
|
-
durationMs?: number;
|
|
1835
|
-
provenance?: OptimizationMethodProvenance;
|
|
1836
|
-
}
|
|
1837
|
-
/**
|
|
1838
|
-
* The durable provenance record. Aligns to the hosted `EvalRunEvent` path but
|
|
1839
|
-
* ADDS the rationale + the explicit baseline→candidate diff (both omitted from
|
|
1840
|
-
* the bare hosted event) + backend provenance.
|
|
1841
|
-
*/
|
|
1842
|
-
interface LoopProvenanceRecord {
|
|
1843
|
-
schema: 'tangle.loop-provenance';
|
|
1844
|
-
/** SHA-256 over the canonical record with this field omitted. */
|
|
1845
|
-
recordDigest: `sha256:${string}`;
|
|
1846
|
-
runId: string;
|
|
1847
|
-
runDir: string;
|
|
1848
|
-
timestamp: string;
|
|
1849
|
-
/** Baseline + winner surface content hashes — distinguishable, byte-verifiable. */
|
|
1850
|
-
baselineContentHash: string;
|
|
1851
|
-
winnerContentHash: string;
|
|
1852
|
-
/** Proposer label/rationale for the promoted change. Absent ⇒ winner == baseline. */
|
|
1853
|
-
winnerLabel?: string;
|
|
1854
|
-
winnerRationale?: string;
|
|
1855
|
-
/** The explicit baseline→winner unified diff the gate decided on. */
|
|
1856
|
-
diff: string;
|
|
1857
|
-
/** Every candidate across every generation, with its rationale and structured cause. */
|
|
1858
|
-
candidates: LoopProvenanceCandidate[];
|
|
1859
|
-
/** Complete external method identity and spend, when one authored the candidate. */
|
|
1860
|
-
optimizationMethod?: LoopProvenanceOptimizationMethod;
|
|
1861
|
-
/** Exact campaign, split, surface, and receipt identities behind every summary. */
|
|
1862
|
-
evidence: LoopProvenanceEvidence;
|
|
1863
|
-
/** Baseline composite on the search split that generated the candidates. */
|
|
1864
|
-
baselineSearchComposite: number;
|
|
1865
|
-
/** The gate verdict — decision + reasons + contributing gates + delta. */
|
|
1866
|
-
gate: {
|
|
1867
|
-
decision: GateDecision;
|
|
1868
|
-
reasons: string[];
|
|
1869
|
-
delta?: number;
|
|
1870
|
-
contributingGates: GateContribution[];
|
|
1871
|
-
};
|
|
1872
|
-
/** Present iff the loop ran with `holdout: 'deferred'` — the held-out
|
|
1873
|
-
* comparison was intentionally not measured in this run, so the holdout
|
|
1874
|
-
* composites and `heldOutLift` are ABSENT rather than recorded as a
|
|
1875
|
-
* meaningless 0. */
|
|
1876
|
-
holdout?: 'deferred';
|
|
1877
|
-
/** baseline-on-holdout composite mean. Absent when `holdout === 'deferred'`. */
|
|
1878
|
-
baselineHoldoutComposite?: number;
|
|
1879
|
-
/** winner-on-holdout composite mean. Absent when `holdout === 'deferred'`. */
|
|
1880
|
-
winnerHoldoutComposite?: number;
|
|
1881
|
-
/** winnerHoldout - baselineHoldout — RECOMPUTABLE from this record. Absent
|
|
1882
|
-
* when `holdout === 'deferred'` (no held-out measurement ran). */
|
|
1883
|
-
heldOutLift?: number;
|
|
1884
|
-
/** Backend provenance: stub-vs-real verdict + worker call count + models. */
|
|
1885
|
-
backend: LoopProvenanceBackend;
|
|
1886
|
-
totalCostUsd: number;
|
|
1887
|
-
totalDurationMs: number;
|
|
1888
|
-
}
|
|
1889
|
-
interface BuildLoopProvenanceArgs<TArtifact, TScenario extends Scenario> {
|
|
1890
|
-
runId: string;
|
|
1891
|
-
runDir: string;
|
|
1892
|
-
timestamp: string;
|
|
1893
|
-
baselineSurface: MutableSurface;
|
|
1894
|
-
winnerSurface: MutableSurface;
|
|
1895
|
-
winnerLabel?: string;
|
|
1896
|
-
winnerRationale?: string;
|
|
1897
|
-
/** Exact baseline campaign on the search split. */
|
|
1898
|
-
baselineSearchCampaign: CampaignResult<TArtifact, TScenario>;
|
|
1899
|
-
/** Per-generation candidate records straight off the loop result. */
|
|
1900
|
-
generations: Array<{
|
|
1901
|
-
generationIndex: number;
|
|
1902
|
-
candidates: GenerationCandidate[];
|
|
1903
|
-
promoted: string[];
|
|
1904
|
-
/** Surfaces measured this generation, keyed by surface hash so the content
|
|
1905
|
-
* hash can be computed and the loop identity rechecked from real bytes. */
|
|
1906
|
-
surfaces: Array<{
|
|
1907
|
-
surfaceHash: string;
|
|
1908
|
-
surface: MutableSurface;
|
|
1909
|
-
campaign: CampaignResult<TArtifact, TScenario>;
|
|
1910
|
-
}>;
|
|
1911
|
-
}>;
|
|
1912
|
-
gate: GateResult;
|
|
1913
|
-
/** Holdout policy the loop ran with. `'deferred'` ⇒ the holdout campaigns
|
|
1914
|
-
* below are the shared empty campaign and the record omits the holdout
|
|
1915
|
-
* composites + `heldOutLift`. Default `'measured'`. */
|
|
1916
|
-
holdout?: 'measured' | 'deferred';
|
|
1917
|
-
baselineOnHoldout: CampaignResult<TArtifact, TScenario>;
|
|
1918
|
-
winnerOnHoldout: CampaignResult<TArtifact, TScenario>;
|
|
1919
|
-
neutralizedSurface?: MutableSurface;
|
|
1920
|
-
neutralizedOnHoldout?: CampaignResult<TArtifact, TScenario>;
|
|
1921
|
-
/** Settled run-wide receipts — agent calls are the source for backend provenance. */
|
|
1922
|
-
costReceipts: ReadonlyArray<CostReceipt>;
|
|
1923
|
-
totalCostUsd: number;
|
|
1924
|
-
totalDurationMs: number;
|
|
1925
|
-
optimizationMethod?: LoopProvenanceOptimizationMethod;
|
|
1926
|
-
}
|
|
1927
|
-
interface LoopProvenanceArgsFromResult<TArtifact, TScenario extends Scenario> {
|
|
1928
|
-
runId: string;
|
|
1929
|
-
runDir: string;
|
|
1930
|
-
timestamp: string;
|
|
1931
|
-
baselineSurface: MutableSurface;
|
|
1932
|
-
result: RunImprovementLoopResult<TArtifact, TScenario>;
|
|
1933
|
-
costReceipts: ReadonlyArray<CostReceipt>;
|
|
1934
|
-
totalCostUsd: number;
|
|
1935
|
-
totalDurationMs: number;
|
|
1936
|
-
}
|
|
1937
|
-
/** One translation from a completed improvement loop into durable evidence. */
|
|
1938
|
-
declare function loopProvenanceArgsFromResult<TArtifact, TScenario extends Scenario>(input: LoopProvenanceArgsFromResult<TArtifact, TScenario>): BuildLoopProvenanceArgs<TArtifact, TScenario>;
|
|
1939
|
-
/** Build the durable provenance record from a completed loop result. */
|
|
1940
|
-
declare function buildLoopProvenanceRecord<TArtifact, TScenario extends Scenario>(args: BuildLoopProvenanceArgs<TArtifact, TScenario>): LoopProvenanceRecord;
|
|
1941
|
-
/** Digest the exact campaign fields that can affect a measured comparison. */
|
|
1942
|
-
declare function campaignMeasurementDigest<TArtifact, TScenario extends Scenario>(campaign: CampaignResult<TArtifact, TScenario>): `sha256:${string}`;
|
|
1943
|
-
/** Recompute and validate the self-addressed durable record. */
|
|
1944
|
-
declare function verifyLoopProvenanceRecord(record: LoopProvenanceRecord): LoopProvenanceRecord;
|
|
1945
|
-
/** SHA-256 over the RFC 8785 canonical JSON of `value`. Throws
|
|
1946
|
-
* `LedgerCanonicalizationError` for a value with no canonical form. */
|
|
1947
|
-
declare function canonicalDigest(value: unknown): `sha256:${string}`;
|
|
1948
|
-
/**
|
|
1949
|
-
* Build the loop's OTLP-ingestable spans from a provenance record. One root
|
|
1950
|
-
* span per loop (`tangle.runId`), one span per generation, one span per
|
|
1951
|
-
* candidate (carrying its surfaceHash + label), and one span for the gate
|
|
1952
|
-
* decision (carrying reasons + delta + lift). Candidate + gate spans pivot on
|
|
1953
|
-
* the same `tangle.runId` / `tangle.generation` attributes `/adapters/otel`
|
|
1954
|
-
* reads, so the hosted collector reconstructs the full tree.
|
|
1955
|
-
*
|
|
1956
|
-
* Times are synthesized monotonically off a single base so the span tree is
|
|
1957
|
-
* orderable; the substrate does not retain per-candidate wall-clock starts.
|
|
1958
|
-
*/
|
|
1959
|
-
declare function loopProvenanceSpans(record: LoopProvenanceRecord, opts?: {
|
|
1960
|
-
baseTimeMs?: number;
|
|
1961
|
-
}): TraceSpanEvent[];
|
|
1962
|
-
/** Canonical durable paths under the run dir. */
|
|
1963
|
-
declare function provenanceRecordPath(runDir: string): string;
|
|
1964
|
-
/**
|
|
1965
|
-
* Canonical path for the durable OTLP spans JSONL file under a loop run directory.
|
|
1966
|
-
*/
|
|
1967
|
-
declare function provenanceSpansPath(runDir: string): string;
|
|
1968
|
-
interface EmitLoopProvenanceResult {
|
|
1969
|
-
record: LoopProvenanceRecord;
|
|
1970
|
-
spans: TraceSpanEvent[];
|
|
1971
|
-
/** Absolute paths the record + spans were written to, when storage persists. */
|
|
1972
|
-
recordPath: string;
|
|
1973
|
-
spansPath: string;
|
|
1974
|
-
}
|
|
1975
|
-
interface EmitLoopProvenanceArgs<TArtifact, TScenario extends Scenario> extends BuildLoopProvenanceArgs<TArtifact, TScenario> {
|
|
1976
|
-
/** Storage the record + spans are written through. */
|
|
1977
|
-
storage: CampaignStorage;
|
|
1978
|
-
/** When set, the spans are also shipped to the hosted `/v1/ingest/traces`
|
|
1979
|
-
* endpoint so the collector receives the full loop, not just `cost.*`. */
|
|
1980
|
-
hostedClient?: HostedClient;
|
|
1981
|
-
}
|
|
1982
|
-
/**
|
|
1983
|
-
* Build the provenance record + OTel spans and persist them durably under the
|
|
1984
|
-
* run dir (and ship spans to a hosted collector when one is wired). Returns
|
|
1985
|
-
* both artifacts so the caller can assert on / re-derive from them.
|
|
1986
|
-
*
|
|
1987
|
-
* Fail-loud: the durable write throws on storage failure (a swallowed write is
|
|
1988
|
-
* exactly the "emitted but lost" failure this closes). The hosted span ship is
|
|
1989
|
-
* the one best-effort leg — its failure is logged, not thrown, so an offline
|
|
1990
|
-
* collector never fails the loop (the durable artifact is the source of truth).
|
|
1991
|
-
*/
|
|
1992
|
-
declare function emitLoopProvenance<TArtifact, TScenario extends Scenario>(args: EmitLoopProvenanceArgs<TArtifact, TScenario>): Promise<EmitLoopProvenanceResult>;
|
|
1993
|
-
//#endregion
|
|
1994
|
-
export { readExternalOptimizerObservationArtifact as $, SearchLedger as $t, CanaryOptions as A, RunEvalOptions as An, SearchHistoryCoverageRow as At, OptimizationMethodResult as B, CacheIssueReason as Bn, OpenSearchLedgerOptions as Bt, RedTeamReport as C, CrowdedFrontierParentOptions as Cn, GepaCandidatePopulationCandidate as Ct, CanaryAlert as D, OpenAutoPrOptions as Dn, CreateSearchHistoryReceiptInput as Dt, scoreRedTeamOutput as E, crowdedFrontierParent as En, readGepaCandidatePopulationArtifact as Et, OptimizationMethod as F, runCampaign as Fn, assertSearchHistoryMatchesReplay as Ft, combineComparisonCosts as G, cellCachePath as Gn, SearchCandidateLineage as Gt, OptimizationMethodScore as H, readCachedCell as Hn, SearchArtifactRef as Ht, OptimizationMethodComparison as I, CampaignRunPlan as In, createSearchHistoryReceipt as It, optimizationTokenUsageFromSummary as J, fsCampaignStorage as Jn, SearchCandidateSlotClosedEvent as Jt, compareOptimizationMethods as K, CampaignStorage as Kn, SearchCandidateRegisteredEvent as Kt, OptimizationMethodInput as L, CampaignRunPlanCell as Ln, searchHistoryCoverageRow as Lt, runCanaries as M, CampaignCellFailureReceipt as Mn, SearchHistoryReceipt as Mt, CompareOptimizationMethodsOptions as N, CampaignCellRetryPolicy as Nn, SearchHistoryRequiredError as Nt, CanaryEvaluation as O, OpenAutoPrResult as On, SearchHistoryAuditSummary as Ot, ComparisonCost as P, RunCampaignOptions as Pn, assertCompleteSearchHistory as Pt, ExternalOptimizerSubmittedCandidate as Q, SearchFailureReason as Qt, OptimizationMethodPairwise as R, PlanCampaignRunOptions as Rn, verifySearchHistoryReceipt as Rt, RedTeamFinding as S, validateSearchLedgerEvent as Sn, GepaCandidatePopulationArtifact as St, redTeamReport as T, ParentSelector as Tn, GepaCandidateSelectionScore as Tt, OptimizationPackageSource as U, CellScheduleSlot as Un, SearchAttemptAccounting as Ut, OptimizationMethodRunOptions as V, CacheRead as Vn, SearchAccountingAudit as Vt, OptimizationTokenUsage as W, buildCellSchedule as Wn, SearchCandidateDecidedEvent as Wt, ExternalOptimizerObservationArtifact as X, SearchCompletedEvent as Xt, ExternalOptimizerExecutionSummary as Y, inMemoryCampaignStorage as Yn, SearchCandidateSurface as Yt, ExternalOptimizerObservationSummary as Z, SearchCostAccounting as Zt, provenanceSpansPath as _, SearchSurfaceKind as _n, SearchLedgerBinding as _t, LoopProvenanceBackend as a, SearchLedgerTrustedHeadMode as an, quotaExhaustedUntil as at, RedTeamCase as b, SearchTokenAccounting as bn, SearchRunIdentity as bt, LoopProvenanceOptimizationMethod as c, SearchOperationRecordedEvent as cn, RunImprovementLoopResult as ct, campaignMeasurementDigest as d, SearchPlannedEvent as dn, RunOptimizationOptions as dt, SearchLedgerAppendResult as en, LlmJudgeDimension as et, canonicalDigest as f, SearchPlannedOperation as fn, RunOptimizationResult as ft, provenanceRecordPath as g, SearchSurfaceEvidence as gn, SearchExecutionIdentity as gt, loopProvenanceSpans as h, SearchSurfaceEffect as hn, ProposedSearchCandidate as ht, LoopProvenanceArgsFromResult as i, SearchLedgerReplay as in, isTransientTransportFailure as it, CanaryReport as j, runEval as jn, SearchHistoryPolicy as jt, CanaryKind as k, openAutoPr as kn, SearchHistoryCoverage as kt, LoopProvenanceRecord as l, SearchPlan as ln, runImprovementLoop as lt, loopProvenanceArgsFromResult as m, SearchSourceRef as mn, MeasuredSearchCandidate as mt, EmitLoopProvenanceArgs as n, SearchLedgerEvent as nn, llmJudge as nt, LoopProvenanceCandidate as o, SearchModelIdentity as on, transientDispatchFailure as ot, emitLoopProvenance as p, SearchPlannedTask as pn, runOptimization as pt, costFromLedgerSummary as q, createRunCostLedger as qn, SearchCandidateSlot as qt, EmitLoopProvenanceResult as r, SearchLedgerHash as rn, TransientFailureOptions as rt, LoopProvenanceEvidence as s, SearchOperationKind as sn, RunImprovementLoopOptions as st, BuildLoopProvenanceArgs as t, SearchLedgerEntry as tn, LlmJudgeOptions as tt, buildLoopProvenanceRecord as u, SearchPlanExtendedEvent as un, PremeasuredOptimizationBaseline as ut, verifyLoopProvenanceRecord as v, SearchTaskAttemptedEvent as vn, SearchRecorder as vt, redTeamDataset as w, ParentSelectionContext as wn, GepaCandidatePopulationSummary as wt, RedTeamCategory as x, openSearchLedger as xn, recordCandidatePopulationSearch as xt, DEFAULT_RED_TEAM_CORPUS as y, SearchTaskOutcome as yn, SearchRecorderOptions as yt, OptimizationMethodProvenance as z, planCampaignRun as zn, FileSearchLedger as zt };
|
|
1995
|
-
//# sourceMappingURL=provenance-CRY67X50.d.ts.map
|