@tangle-network/agent-eval 0.173.2 → 0.174.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +40 -0
- package/dist/{proposal-findings-bko3GGy-.js → abort-signal-CtzAM_sJ.js} +11 -11
- package/dist/abort-signal-CtzAM_sJ.js.map +1 -0
- package/dist/adapters/http.d.ts +2 -2
- package/dist/agent-profile-_xPxqVJt.d.ts +488 -0
- package/dist/agent-profile-_xPxqVJt.d.ts.map +1 -0
- package/dist/analyst/index.d.ts +7 -9
- package/dist/analyst/index.d.ts.map +1 -1
- package/dist/analyst/index.js +8 -8
- package/dist/{benchmark-C4wk_Sjr.js → benchmark-DQKzykkO.js} +2 -2
- package/dist/{benchmark-C4wk_Sjr.js.map → benchmark-DQKzykkO.js.map} +1 -1
- package/dist/{benchmark-command-CS6gVHVq.js → benchmark-command-mZIlR-ra.js} +13 -13
- package/dist/{benchmark-command-CS6gVHVq.js.map → benchmark-command-mZIlR-ra.js.map} +1 -1
- package/dist/benchmarks/index.d.ts +3 -4
- package/dist/benchmarks/index.d.ts.map +1 -1
- package/dist/benchmarks/index.js +3 -3
- package/dist/campaign/index.d.ts +5 -9
- package/dist/campaign/index.js +7 -7
- package/dist/{campaign-B3kPMU8S.js → campaign-BzMSCejE.js} +8 -8
- package/dist/{campaign-B3kPMU8S.js.map → campaign-BzMSCejE.js.map} +1 -1
- package/dist/cli.js +1 -1
- package/dist/{client-DlqdbM7n.d.ts → client-vyYQg3bm.d.ts} +2 -2
- package/dist/{client-DlqdbM7n.d.ts.map → client-vyYQg3bm.d.ts.map} +1 -1
- package/dist/contract/index.d.ts +9 -10
- package/dist/contract/index.js +8 -8
- package/dist/{default-registry-B0bKikCb.js → default-registry-CrAp0pYq.js} +4 -4
- package/dist/{default-registry-B0bKikCb.js.map → default-registry-CrAp0pYq.js.map} +1 -1
- package/dist/{default-registry-BKwc8bN5.d.ts → default-registry-FfNzaUHV.d.ts} +3 -3
- package/dist/{default-registry-BKwc8bN5.d.ts.map → default-registry-FfNzaUHV.d.ts.map} +1 -1
- package/dist/{define-agent-eval-CY6qdlGV.d.ts → define-agent-eval-V1jQyCDR.d.ts} +102 -11
- package/dist/define-agent-eval-V1jQyCDR.d.ts.map +1 -0
- package/dist/{define-agent-eval-8h3lXXee.js → define-agent-eval-ox5McL6e.js} +331 -144
- package/dist/define-agent-eval-ox5McL6e.js.map +1 -0
- package/dist/{dspy-rlm-engine-CF0t2ITD.js → dspy-rlm-engine-Caz2pl4L.js} +3 -3
- package/dist/{dspy-rlm-engine-CF0t2ITD.js.map → dspy-rlm-engine-Caz2pl4L.js.map} +1 -1
- package/dist/{engine-DhFir3Ys.d.ts → engine-CvW_I72-.d.ts} +2 -2
- package/dist/{engine-DhFir3Ys.d.ts.map → engine-CvW_I72-.d.ts.map} +1 -1
- package/dist/experiment/index.d.ts +1 -4
- package/dist/experiment/index.d.ts.map +1 -1
- package/dist/{external-optimizer-process-BwITA9Jp.js → external-optimizer-process-CxnFL1hd.js} +2 -2
- package/dist/{external-optimizer-process-BwITA9Jp.js.map → external-optimizer-process-CxnFL1hd.js.map} +1 -1
- package/dist/{external-optimizer-subprocess-wBWeoG6A.js → external-optimizer-subprocess-CQi27uEI.js} +2 -2
- package/dist/{external-optimizer-subprocess-wBWeoG6A.js.map → external-optimizer-subprocess-CQi27uEI.js.map} +1 -1
- package/dist/fuzz.js +1 -1
- package/dist/fuzz.js.map +1 -1
- package/dist/hosted/index.d.ts +1 -1
- package/dist/{index-BQqOjerE.d.ts → index-BTrx5s8m.d.ts} +8 -9
- package/dist/index-BTrx5s8m.d.ts.map +1 -0
- package/dist/{index-D0Db5X-4.d.ts → index-Bn-nlnSV.d.ts} +4 -4
- package/dist/{index-D0Db5X-4.d.ts.map → index-Bn-nlnSV.d.ts.map} +1 -1
- package/dist/index-DKXuBPXf.d.ts +3840 -0
- package/dist/index-DKXuBPXf.d.ts.map +1 -0
- package/dist/index.d.ts +11 -13
- package/dist/index.d.ts.map +1 -1
- package/dist/index.js +10 -10
- package/dist/{kind-factory-gP6lDySe.js → kind-factory-BLvL-E44.js} +2 -2
- package/dist/{kind-factory-gP6lDySe.js.map → kind-factory-BLvL-E44.js.map} +1 -1
- package/dist/{llm-judge-BfqMFo4h.js → llm-judge-DmNaBrXB.js} +2541 -2435
- package/dist/llm-judge-DmNaBrXB.js.map +1 -0
- package/dist/{matrix-DGu8KhSs.d.ts → matrix-CJtXz1ky.d.ts} +2 -2
- package/dist/{matrix-DGu8KhSs.d.ts.map → matrix-CJtXz1ky.d.ts.map} +1 -1
- package/dist/multishot/golden/index.d.ts +1 -1
- package/dist/multishot/index.d.ts +2 -2
- package/dist/openapi.json +1 -1
- package/dist/{produced-state-D91uDvQw.js → produced-state-B8mw6zj9.js} +2 -2
- package/dist/{produced-state-D91uDvQw.js.map → produced-state-B8mw6zj9.js.map} +1 -1
- package/dist/rl.d.ts +1 -1
- package/dist/rl.d.ts.map +1 -1
- package/dist/rl.js.map +1 -1
- package/dist/{semantic-concept-judge-Dok7_35a.js → semantic-concept-judge-E3s_fEjB.js} +3 -3
- package/dist/{semantic-concept-judge-Dok7_35a.js.map → semantic-concept-judge-E3s_fEjB.js.map} +1 -1
- package/dist/{skillopt-optimization-method-DDw3v3gA.js → skillopt-optimization-method-f7399oGb.js} +5 -5
- package/dist/{skillopt-optimization-method-DDw3v3gA.js.map → skillopt-optimization-method-f7399oGb.js.map} +1 -1
- package/dist/statistical-heldout-Cqb73yE9.d.ts +1127 -0
- package/dist/statistical-heldout-Cqb73yE9.d.ts.map +1 -0
- package/dist/{store-otlp-Dow0pk_5.js → store-otlp-DV_H2HDu.js} +2 -2
- package/dist/{store-otlp-Dow0pk_5.js.map → store-otlp-DV_H2HDu.js.map} +1 -1
- package/dist/{store-tool-spans-CCZNsihA.d.ts → store-tool-spans-4o55ABER.d.ts} +3 -3
- package/dist/{store-tool-spans-CCZNsihA.d.ts.map → store-tool-spans-4o55ABER.d.ts.map} +1 -1
- package/dist/{store-tool-spans-CeNj_m2L.js → store-tool-spans-B9tjys_h.js} +3 -3
- package/dist/{store-tool-spans-CeNj_m2L.js.map → store-tool-spans-B9tjys_h.js.map} +1 -1
- package/dist/supervisor-run/index.d.ts.map +1 -1
- package/dist/supervisor-run/index.js +25 -7
- package/dist/supervisor-run/index.js.map +1 -1
- package/dist/{task-failure-attributes-CZjZeBsY.js → task-failure-attributes-CUy9mkIY.js} +2 -2
- package/dist/{task-failure-attributes-CZjZeBsY.js.map → task-failure-attributes-CUy9mkIY.js.map} +1 -1
- package/dist/{tool-groups-Cp4Xdzrp.d.ts → tool-groups-DAe1t6zb.d.ts} +2 -2
- package/dist/tool-groups-DAe1t6zb.d.ts.map +1 -0
- package/dist/trace-repair/index.d.ts +1 -1
- package/dist/traces.d.ts +2 -2
- package/dist/traces.js +4 -4
- package/dist/{types-Ba5UQyVD.d.ts → types-BJz2CPTM.d.ts} +2 -2
- package/dist/{types-Ba5UQyVD.d.ts.map → types-BJz2CPTM.d.ts.map} +1 -1
- package/dist/{types-CiWITkGo.js → types-DQ0e2E7y.js} +2 -2
- package/dist/types-DQ0e2E7y.js.map +1 -0
- package/dist/{types-BDV4PiMR.d.ts → types-Dd1ejaeI.d.ts} +2 -2
- package/dist/{types-BDV4PiMR.d.ts.map → types-Dd1ejaeI.d.ts.map} +1 -1
- package/docs/campaign-proposers.md +42 -0
- package/package.json +1 -1
- package/dist/agent-profile-B9_GGsG8.d.ts +0 -84
- package/dist/agent-profile-B9_GGsG8.d.ts.map +0 -1
- package/dist/backend-integrity-CeuTgqsd.d.ts +0 -280
- package/dist/backend-integrity-CeuTgqsd.d.ts.map +0 -1
- package/dist/benchmark-BjLGkfnN.d.ts +0 -236
- package/dist/benchmark-BjLGkfnN.d.ts.map +0 -1
- package/dist/define-agent-eval-8h3lXXee.js.map +0 -1
- package/dist/define-agent-eval-CY6qdlGV.d.ts.map +0 -1
- package/dist/external-optimizer-contracts-CQCpyrIL.d.ts +0 -172
- package/dist/external-optimizer-contracts-CQCpyrIL.d.ts.map +0 -1
- package/dist/heldout-gate-Df5hsqmm.d.ts +0 -453
- package/dist/heldout-gate-Df5hsqmm.d.ts.map +0 -1
- package/dist/index-BQqOjerE.d.ts.map +0 -1
- package/dist/index-CFDffsKz.d.ts +0 -1135
- package/dist/index-CFDffsKz.d.ts.map +0 -1
- package/dist/llm-judge-BfqMFo4h.js.map +0 -1
- package/dist/power-preflight-Ptse_Kq7.d.ts +0 -117
- package/dist/power-preflight-Ptse_Kq7.d.ts.map +0 -1
- package/dist/pre-registration-BoI4ucR3.d.ts +0 -592
- package/dist/pre-registration-BoI4ucR3.d.ts.map +0 -1
- package/dist/promotion-policy-CvMda3kU.d.ts +0 -134
- package/dist/promotion-policy-CvMda3kU.d.ts.map +0 -1
- package/dist/proposal-findings-bko3GGy-.js.map +0 -1
- package/dist/provenance-CRY67X50.d.ts +0 -1995
- package/dist/provenance-CRY67X50.d.ts.map +0 -1
- package/dist/statistical-heldout-DTyB_6-1.d.ts +0 -295
- package/dist/statistical-heldout-DTyB_6-1.d.ts.map +0 -1
- package/dist/tool-groups-Cp4Xdzrp.d.ts.map +0 -1
- package/dist/types-CiWITkGo.js.map +0 -1
|
@@ -0,0 +1,3840 @@
|
|
|
1
|
+
import { c as ValidationError, t as AgentEvalError } from "./errors-DEE6u6ot.js";
|
|
2
|
+
import { T as RunPaidCallInput, c as CostLedgerHandle, f as CostLedgerSummary, m as CostReceipt, o as CostLedger, p as CostProvenance } from "./cost-ledger-DbQdN3nO.js";
|
|
3
|
+
import { a as RunRecord, s as RunSplitTag } from "./run-record-DTv1MdjK.js";
|
|
4
|
+
import { _ as ProposalFinding, f as AnalystUsageReceipt, i as AnalystFinding, l as AnalystRunResult, w as TraceAnalysisStore } from "./types-DN2WdT5S.js";
|
|
5
|
+
import { Y as RawProviderSink, it as ServedModelPolicy, p as ChatClient, w as LlmCallMetadata } from "./types-gvRsyJLh.js";
|
|
6
|
+
import { P as PairedArmsComparison } from "./statistical-heldout-Cqb73yE9.js";
|
|
7
|
+
import { d as PairedBootstrapResult } from "./paired-promotion-decision-CGzg0cI_.js";
|
|
8
|
+
import { A as LabeledScenarioWrite, B as ScoredSurfaceOutcome, C as JudgeDimension, D as LabeledScenarioSampleArgs, E as LabeledScenarioRecord, H as SurfaceProposer, N as ParetoParent, O as LabeledScenarioSource, R as Scenario, S as JudgeConfig, T as LabelTrust, _ as GateDecision, a as CampaignResult, b as GenerationRecord, c as CampaignTraceWriter, d as DispatchContext, f as DispatchFn, g as GateContribution, i as CampaignCostMeter, j as MutableSurface, k as LabeledScenarioStore, l as CodeSurface, o as CampaignScenarioIdentity, p as Gate, r as CampaignCellResult, u as ComponentSurface, v as GateResult, y as GenerationCandidate } from "./types-BJz2CPTM.js";
|
|
9
|
+
import { m as LedgerTrustedHeadRemoval, p as LedgerTrustedHead } from "./index-D-UdhAmg.js";
|
|
10
|
+
import { r as LedgerHash } from "./canonical-CFpojCN5.js";
|
|
11
|
+
import { B as ExternalOptimizerModelBudget, E as AnalystIssueExpectation, F as ExternalOptimizerCallbackLimits, J as ExternalOptimizerWireCounts, K as ExternalOptimizerResumeMode, R as ExternalOptimizerEvaluationObservation, V as ExternalOptimizerModelCall, X as ExternalTextEvaluationRequest, Y as ExternalTextCandidate, m as AnalystBenchmarkLabelState, q as ExternalOptimizerRunnerCommand, t as AgentProfile$1, u as AnalystBenchmarkCase } from "./agent-profile-_xPxqVJt.js";
|
|
12
|
+
import { r as DatasetScenario, t as Dataset } from "./dataset-DQqhOCPt.js";
|
|
13
|
+
import { a as CheckerIdentity, n as VerdictCertification, t as DefaultVerdict, x as VerificationStrategySource } from "./verdict-E4eRNf7-.js";
|
|
14
|
+
import { t as DetectRewardHackingInput } from "./reward-hacking-ZXEi9VCq.js";
|
|
15
|
+
import { g as TraceSpanEvent, t as HostedClient } from "./client-vyYQg3bm.js";
|
|
16
|
+
import { AgentProfile } from "@tangle-network/agent-interface";
|
|
17
|
+
import { z } from "zod";
|
|
18
|
+
//#region src/campaign/coverage.d.ts
|
|
19
|
+
/** Reject campaign designs whose denominator cannot be identified exactly. */
|
|
20
|
+
declare function assertCampaignDesign<TScenario extends Scenario>(scenarios: readonly TScenario[], reps: number): void;
|
|
21
|
+
/** Redacted but independently verifiable identity of one complete scenario. */
|
|
22
|
+
declare function campaignScenarioIdentity<TScenario extends Scenario>(scenario: TScenario): CampaignScenarioIdentity & Pick<TScenario, 'id' | 'kind'>;
|
|
23
|
+
/** Canonical split identity reconstructed from redacted scenario identities. */
|
|
24
|
+
declare function campaignSplitDigestFromIdentities(scenarios: readonly CampaignScenarioIdentity[], reps: number): `sha256:${string}`;
|
|
25
|
+
/** Canonical identity of the exact scenario payloads and replicate count. */
|
|
26
|
+
declare function campaignSplitDigest<TScenario extends Scenario>(scenarios: readonly TScenario[], reps: number): `sha256:${string}`;
|
|
27
|
+
/** Refuse a campaign whose retained task identities contradict its split digest. */
|
|
28
|
+
declare function assertCampaignSplitIdentity(scenarios: readonly CampaignScenarioIdentity[], reps: number, splitDigest: string): void;
|
|
29
|
+
//#endregion
|
|
30
|
+
//#region src/campaign/storage.d.ts
|
|
31
|
+
/**
|
|
32
|
+
* `CampaignStorage` — the filesystem seam `runCampaign` writes through
|
|
33
|
+
* (run/cell dirs, the resumability cache, per-cell artifacts, trace spans).
|
|
34
|
+
*
|
|
35
|
+
* The default (`fsCampaignStorage`) is the Node filesystem — identical
|
|
36
|
+
* behavior to the inline `node:fs` calls it replaces, so existing CLI
|
|
37
|
+
* consumers are unaffected. `inMemoryCampaignStorage` keeps everything in a
|
|
38
|
+
* `Map`, so the substrate runs in environments WITHOUT a filesystem
|
|
39
|
+
* (Cloudflare Workers, Deno Deploy, other edge runtimes) — the campaign
|
|
40
|
+
* still produces its `CampaignResult` (cells + aggregates) in memory;
|
|
41
|
+
* artifacts/traces simply aren't persisted to disk.
|
|
42
|
+
*
|
|
43
|
+
* Paths are opaque keys to the in-memory adapter — it does not parse them,
|
|
44
|
+
* so the same `join(...)`-built paths work unchanged across both adapters.
|
|
45
|
+
*/
|
|
46
|
+
interface CampaignStorage {
|
|
47
|
+
/** Ensure a directory exists (recursive). No-op for in-memory. */
|
|
48
|
+
ensureDir(dir: string): void;
|
|
49
|
+
/** Does this path exist (as a written file or an ensured dir)? */
|
|
50
|
+
exists(path: string): boolean;
|
|
51
|
+
/** Read a UTF-8 file; `undefined` when missing or unreadable. */
|
|
52
|
+
read(path: string): string | undefined;
|
|
53
|
+
/** Write a file (string or bytes). Parent dir is assumed ensured. */
|
|
54
|
+
write(path: string, content: string | Uint8Array): void;
|
|
55
|
+
/** Append only when the current UTF-8 byte length matches `expectedBytes`.
|
|
56
|
+
* Returns the new length, or undefined when another writer won. */
|
|
57
|
+
append(path: string, content: string, expectedBytes: number): number | undefined;
|
|
58
|
+
}
|
|
59
|
+
/** Node-filesystem storage — the default. Lazily requires `node:fs` so the
|
|
60
|
+
* module imports cleanly in non-Node runtimes (where the caller passes
|
|
61
|
+
* `inMemoryCampaignStorage` instead and never constructs this).
|
|
62
|
+
*
|
|
63
|
+
* `createRequire(import.meta.url)` is the ESM-native lazy require — a bare
|
|
64
|
+
* `require` is a ReferenceError under `"type": "module"`, which is exactly
|
|
65
|
+
* the shape this package publishes. */
|
|
66
|
+
declare function fsCampaignStorage(): CampaignStorage;
|
|
67
|
+
/** In-memory storage for filesystem-less runtimes. Artifacts + trace spans
|
|
68
|
+
* live in a `Map` for the duration of the run; the `CampaignResult` is
|
|
69
|
+
* fully populated, but nothing is persisted to disk. */
|
|
70
|
+
declare function inMemoryCampaignStorage(): CampaignStorage;
|
|
71
|
+
/** Open the durable spend account stored beside a logical run. */
|
|
72
|
+
declare function createRunCostLedger(input: {
|
|
73
|
+
storage: CampaignStorage;
|
|
74
|
+
runDir: string;
|
|
75
|
+
costCeilingUsd?: number;
|
|
76
|
+
/** Set false for read-only inspection that must not create the run directory. */
|
|
77
|
+
ensureRunDir?: boolean;
|
|
78
|
+
}): CostLedger;
|
|
79
|
+
//#endregion
|
|
80
|
+
//#region src/campaign/cell-schedule.d.ts
|
|
81
|
+
declare function buildCellSchedule<TScenario extends Scenario>(scenarios: TScenario[], seed: number, reps: number): Array<{
|
|
82
|
+
scenario: TScenario;
|
|
83
|
+
rep: number;
|
|
84
|
+
cellId: string;
|
|
85
|
+
cellSeed: number;
|
|
86
|
+
}>;
|
|
87
|
+
type CellScheduleSlot<TScenario extends Scenario> = ReturnType<typeof buildCellSchedule<TScenario>>[number];
|
|
88
|
+
declare function cellCachePath(runDir: string, cellId: string): string;
|
|
89
|
+
//#endregion
|
|
90
|
+
//#region src/campaign/cell-cache.d.ts
|
|
91
|
+
type CacheIssueReason = 'missing' | 'manifest-mismatch' | 'cell-mismatch' | 'missing-cost-provenance' | 'invalid-cost-provenance' | 'invalid-cost-receipts' | 'corrupt';
|
|
92
|
+
type CacheRead<TArtifact> = {
|
|
93
|
+
status: 'hit';
|
|
94
|
+
cell: CampaignCellResult<TArtifact>;
|
|
95
|
+
} | {
|
|
96
|
+
status: 'miss';
|
|
97
|
+
reason: CacheIssueReason;
|
|
98
|
+
};
|
|
99
|
+
declare function readCachedCell<TArtifact>(args: {
|
|
100
|
+
storage: CampaignStorage;
|
|
101
|
+
cachePath: string;
|
|
102
|
+
cellId: string;
|
|
103
|
+
manifestHash: string;
|
|
104
|
+
}): CacheRead<TArtifact>;
|
|
105
|
+
//#endregion
|
|
106
|
+
//#region src/campaign/plan-campaign-run.d.ts
|
|
107
|
+
interface CampaignRunPlanCell {
|
|
108
|
+
cellId: string;
|
|
109
|
+
scenarioId: string;
|
|
110
|
+
rep: number;
|
|
111
|
+
seed: number;
|
|
112
|
+
cachePath: string;
|
|
113
|
+
status: 'cached' | 'run' | 'blocked';
|
|
114
|
+
reason?: CacheIssueReason | 'resumable-off';
|
|
115
|
+
}
|
|
116
|
+
interface CampaignRunPlan {
|
|
117
|
+
manifestHash: string;
|
|
118
|
+
splitDigest: `sha256:${string}`;
|
|
119
|
+
totalCells: number;
|
|
120
|
+
cellsCached: number;
|
|
121
|
+
cellsBlocked: number;
|
|
122
|
+
cellsToRun: number;
|
|
123
|
+
cells: CampaignRunPlanCell[];
|
|
124
|
+
}
|
|
125
|
+
interface PlanCampaignRunOptions<TScenario extends Scenario, TArtifact> {
|
|
126
|
+
scenarios: TScenario[];
|
|
127
|
+
dispatch?: DispatchFn<TScenario, TArtifact>;
|
|
128
|
+
dispatchRef?: string;
|
|
129
|
+
judges?: JudgeConfig<TArtifact, TScenario>[];
|
|
130
|
+
seed?: number;
|
|
131
|
+
reps?: number;
|
|
132
|
+
resumable?: boolean;
|
|
133
|
+
/** See RunCampaignOptions.rerunInvalidCachedCells. */
|
|
134
|
+
rerunInvalidCachedCells?: boolean;
|
|
135
|
+
runDir: string;
|
|
136
|
+
/** Subject repo for the shared run-dir root (see RunCampaignOptions.repo). */
|
|
137
|
+
repo?: string;
|
|
138
|
+
storage?: CampaignStorage;
|
|
139
|
+
/** Spend account used to validate cached receipt identities. */
|
|
140
|
+
costLedger?: CostLedgerHandle;
|
|
141
|
+
/** Receipt tags used by the campaign that produced the cached cells. */
|
|
142
|
+
costTags?: Readonly<Record<string, string>>;
|
|
143
|
+
}
|
|
144
|
+
/**
|
|
145
|
+
* Plan a campaign WITHOUT dispatching: computes the manifest hash and the per-cell
|
|
146
|
+
* run-vs-cached schedule so callers can preview cost and resumability before spending.
|
|
147
|
+
*/
|
|
148
|
+
declare function planCampaignRun<TScenario extends Scenario, TArtifact>(opts: PlanCampaignRunOptions<TScenario, TArtifact>): CampaignRunPlan;
|
|
149
|
+
//#endregion
|
|
150
|
+
//#region src/campaign/run-campaign.d.ts
|
|
151
|
+
interface RunCampaignOptions<TScenario extends Scenario, TArtifact> {
|
|
152
|
+
scenarios: TScenario[];
|
|
153
|
+
dispatch: DispatchFn<TScenario, TArtifact>;
|
|
154
|
+
/** Abort active dispatches when the owning operation is cancelled. */
|
|
155
|
+
signal?: AbortSignal;
|
|
156
|
+
/**
|
|
157
|
+
* Stable identity for the dispatch behavior, included in the manifest/cache
|
|
158
|
+
* key. Set this when the same function name can run different models,
|
|
159
|
+
* prompts, tools, or external config.
|
|
160
|
+
*/
|
|
161
|
+
dispatchRef?: string;
|
|
162
|
+
judges?: JudgeConfig<TArtifact, TScenario>[];
|
|
163
|
+
/** Required for reproducibility. Default 42. */
|
|
164
|
+
seed?: number;
|
|
165
|
+
/** Per-scenario replicates for CI bands. Default 1; raise to 5+ for
|
|
166
|
+
* bootstrap-tight intervals on critical eval. */
|
|
167
|
+
reps?: number;
|
|
168
|
+
/** When true (default), completed cells are cached by
|
|
169
|
+
* (manifestHash, scenarioId, rep, generation). Re-runs skip cached cells. */
|
|
170
|
+
resumable?: boolean;
|
|
171
|
+
/**
|
|
172
|
+
* Optional explicit cell selection. The campaign manifest and split digest
|
|
173
|
+
* still describe the complete declared scenario × replicate design; this
|
|
174
|
+
* only limits the rows executed by this invocation.
|
|
175
|
+
*/
|
|
176
|
+
cellFilter?: (input: {
|
|
177
|
+
scenario: TScenario;
|
|
178
|
+
rep: number;
|
|
179
|
+
}) => boolean;
|
|
180
|
+
/**
|
|
181
|
+
* Reuse a cached cell that has an error instead of dispatching it again.
|
|
182
|
+
* The default retries failed cells, preserving normal campaign behaviour.
|
|
183
|
+
*/
|
|
184
|
+
reuseFailedCells?: boolean;
|
|
185
|
+
/**
|
|
186
|
+
* Explicitly rerun only cached cells whose saved result is unreadable or
|
|
187
|
+
* has missing/invalid cost provenance. Valid cached cells remain reusable.
|
|
188
|
+
* Default false refuses to begin work when any such cache entry exists.
|
|
189
|
+
*/
|
|
190
|
+
rerunInvalidCachedCells?: boolean;
|
|
191
|
+
/** Optional store — when present, every artifact + judge score is captured
|
|
192
|
+
* with the configured `captureSource`. Capture is default ON; pass `'off'`
|
|
193
|
+
* to disable. */
|
|
194
|
+
labeledStore?: LabeledScenarioStore | 'off';
|
|
195
|
+
captureSource?: 'production-trace' | 'eval-run' | 'manual' | 'red-team' | 'synthetic';
|
|
196
|
+
captureSourceVersionHash?: string;
|
|
197
|
+
/** Hard spend cap. Each paid call reserves its enforced maximum before dispatch. */
|
|
198
|
+
costCeiling?: number;
|
|
199
|
+
/** Shared spend account. Improvement loops pass one ledger through every
|
|
200
|
+
* campaign so the ceiling and returned total are run-wide. */
|
|
201
|
+
costLedger?: CostLedgerHandle;
|
|
202
|
+
/** Attribution label for receipts recorded by this campaign. */
|
|
203
|
+
costPhase?: string;
|
|
204
|
+
/** Additional immutable receipt tags supplied by an owning workflow. */
|
|
205
|
+
costTags?: Readonly<Record<string, string>>;
|
|
206
|
+
/** Max concurrent cells. Default 2. */
|
|
207
|
+
maxConcurrency?: number;
|
|
208
|
+
/**
|
|
209
|
+
* Stop after the first dispatch or judge error. The failed cell is persisted
|
|
210
|
+
* before active sibling cells are aborted and drained, then the campaign
|
|
211
|
+
* rejects with the exact error thrown by that dispatch or judge.
|
|
212
|
+
* Default false preserves the normal behavior of returning failed cells and
|
|
213
|
+
* continuing the remaining schedule.
|
|
214
|
+
* With `cellRetry`, a retryable failure is not an error yet: this abort
|
|
215
|
+
* fires only when a cell's final attempt fails.
|
|
216
|
+
*/
|
|
217
|
+
abortOnCellError?: boolean;
|
|
218
|
+
/**
|
|
219
|
+
* Opt-in bounded in-run retry of a failed cell. Absent by default: a failed
|
|
220
|
+
* cell is final on its first attempt. A retried attempt re-runs the SAME
|
|
221
|
+
* slot — same `cellId`, same `seed`, same cost tags — so the schedule,
|
|
222
|
+
* manifest, and pairing are unchanged. An attempt that failed because the
|
|
223
|
+
* campaign was cancelled is never retried.
|
|
224
|
+
*/
|
|
225
|
+
cellRetry?: CampaignCellRetryPolicy;
|
|
226
|
+
/**
|
|
227
|
+
* Per-cell dispatch deadline in ms. A `dispatch` that neither resolves nor
|
|
228
|
+
* rejects within this window is a hang (a stalled model request, an
|
|
229
|
+
* exhausted runtime resource, a backend that never closes its stream). When
|
|
230
|
+
* set, the cell's `ctx.signal` is aborted. A dispatch that stops is recorded
|
|
231
|
+
* as an error (`dispatch exceeded <N>ms`). A dispatch that ignores
|
|
232
|
+
* cancellation rejects the campaign without publishing incomplete cost data.
|
|
233
|
+
* `undefined`/`0` means unbounded.
|
|
234
|
+
*/
|
|
235
|
+
dispatchTimeoutMs?: number;
|
|
236
|
+
/**
|
|
237
|
+
* Time allowed for an aborted dispatch and its paid calls to stop before the
|
|
238
|
+
* campaign rejects without producing a result. Default 5 seconds.
|
|
239
|
+
*/
|
|
240
|
+
dispatchShutdownTimeoutMs?: number;
|
|
241
|
+
/** Required: where artifacts + traces land. A bare name (not an absolute path)
|
|
242
|
+
* resolves to the shared `~/.tangle/traces/<repo>/runs/<name>` root so run
|
|
243
|
+
* bundles never pollute a repo working tree. Pass an absolute path to override. */
|
|
244
|
+
runDir: string;
|
|
245
|
+
/** Subject repo for the shared run-dir root (defaults to the CWD basename).
|
|
246
|
+
* Only consulted when `runDir` is a bare name. */
|
|
247
|
+
repo?: string;
|
|
248
|
+
/** Tracing posture. Default is the substrate's `FileSystemTraceStore` rooted
|
|
249
|
+
* at `<runDir>/traces/`. `'off'` disables capture entirely — substrate
|
|
250
|
+
* refuses this when the caller wires `autoOnPromote !== 'none'`. */
|
|
251
|
+
tracing?: 'on' | 'off';
|
|
252
|
+
/**
|
|
253
|
+
* Per-cell usage expectation — the early, fine-grained sibling of the
|
|
254
|
+
* batch `assertRealBackend` guard. A cell that produced an artifact (no
|
|
255
|
+
* error) but reported `costUsd === 0` AND zero tokens is a stub: the
|
|
256
|
+
* dispatch never reported LLM activity via `ctx.cost`. Modes:
|
|
257
|
+
* - `'warn'` (default) — log the offending cell loudly, keep going.
|
|
258
|
+
* - `'assert'` — throw `BackendIntegrityError` on the first such cell
|
|
259
|
+
* (fail-fast; recommended for CI campaigns expecting real LLM calls).
|
|
260
|
+
* - `'off'` — no check (replay / deterministic-only / offline analysis).
|
|
261
|
+
*/
|
|
262
|
+
expectUsage?: 'assert' | 'warn' | 'off';
|
|
263
|
+
/** Test seam — override the wall clock for deterministic tests. */
|
|
264
|
+
now?: () => Date;
|
|
265
|
+
/** Test seam — override per-cell trace writer factory. */
|
|
266
|
+
buildTraceWriter?: (cellId: string, dir: string) => CampaignTraceWriter;
|
|
267
|
+
/** Storage backend for run/cell dirs, the resumability cache, artifacts,
|
|
268
|
+
* and trace spans. Default: the Node filesystem (`fsCampaignStorage`).
|
|
269
|
+
* Pass `inMemoryCampaignStorage()` to run in a filesystem-less runtime
|
|
270
|
+
* (Cloudflare Workers, Deno, edge) — the `CampaignResult` is still
|
|
271
|
+
* produced; artifacts/traces just aren't persisted to disk. */
|
|
272
|
+
storage?: CampaignStorage;
|
|
273
|
+
/**
|
|
274
|
+
* Optional per-cell placement strategy. Returns an opaque string the
|
|
275
|
+
* substrate forwards as `ctx.placement` to the Dispatch — placement-aware
|
|
276
|
+
* Dispatches (e.g. `httpDispatch` from `/adapters/http`) use it to route
|
|
277
|
+
* each cell to the right worker, region, or sandbox. When unset, every
|
|
278
|
+
* cell receives `ctx.placement = undefined` and behaves identically to
|
|
279
|
+
* the in-process case.
|
|
280
|
+
*
|
|
281
|
+
* @example
|
|
282
|
+
* cellPlacement: ({ scenario }) => scenario.tags?.includes('eu') ? 'eu-west' : 'us-east'
|
|
283
|
+
*/
|
|
284
|
+
cellPlacement?: (input: {
|
|
285
|
+
scenario: TScenario;
|
|
286
|
+
rep: number;
|
|
287
|
+
generation?: number;
|
|
288
|
+
}) => string | undefined;
|
|
289
|
+
}
|
|
290
|
+
/** Durable `<cell>/failure-receipt.json` written before a failed cell can
|
|
291
|
+
* trigger campaign-wide cancellation. The cell records dispatch measurements;
|
|
292
|
+
* `cost` covers every settled agent and judge call attributed to this exact run
|
|
293
|
+
* attempt. */
|
|
294
|
+
interface CampaignCellFailureReceipt<TArtifact = unknown> {
|
|
295
|
+
schemaVersion: 1;
|
|
296
|
+
runAttemptId: string;
|
|
297
|
+
recordedAt: string;
|
|
298
|
+
failure: {
|
|
299
|
+
stage: 'dispatch' | 'judge';
|
|
300
|
+
judge?: string;
|
|
301
|
+
error: {
|
|
302
|
+
name: string;
|
|
303
|
+
message: string;
|
|
304
|
+
stack?: string;
|
|
305
|
+
};
|
|
306
|
+
};
|
|
307
|
+
cell: CampaignCellResult<TArtifact>;
|
|
308
|
+
cost: CostLedgerSummary;
|
|
309
|
+
}
|
|
310
|
+
/**
|
|
311
|
+
* Bounded in-run retry of failed cells. Every attempt dispatches the same
|
|
312
|
+
* slot and charges the shared cost ledger, so the final cell's `costUsd`,
|
|
313
|
+
* `tokenUsage`, and `costCallIds` cover all attempts. Each retried attempt
|
|
314
|
+
* keeps its failure receipt at `<cell>/failure-receipt.attempt-<n>.json`; a
|
|
315
|
+
* final failed attempt keeps the usual `<cell>/failure-receipt.json`. The
|
|
316
|
+
* final cell records the retry count as `retryAttempts`. Artifacts and trace
|
|
317
|
+
* spans written by a later attempt replace those of the retried attempt; the
|
|
318
|
+
* per-attempt failure receipts are the durable evidence.
|
|
319
|
+
*/
|
|
320
|
+
interface CampaignCellRetryPolicy {
|
|
321
|
+
/** Total attempts per cell, including the first. A positive safe integer. */
|
|
322
|
+
attempts: number;
|
|
323
|
+
/** Decides whether a failed attempt is dispatched again. Receives the
|
|
324
|
+
* receipt's `failure` record. `transientDispatchFailure()` is the
|
|
325
|
+
* ready-made predicate for infrastructure hiccups (502/503/504, dropped
|
|
326
|
+
* streams, admission rejections). */
|
|
327
|
+
retryable: (failure: CampaignCellFailureReceipt['failure']) => boolean;
|
|
328
|
+
}
|
|
329
|
+
/**
|
|
330
|
+
* Core campaign orchestrator: fan scenarios through dispatch, score with judges, aggregate bootstrap CIs, and persist reproducible `CampaignResult` records.
|
|
331
|
+
*/
|
|
332
|
+
declare function runCampaign<TScenario extends Scenario, TArtifact>(opts: RunCampaignOptions<TScenario, TArtifact>): Promise<CampaignResult<TArtifact, TScenario>>;
|
|
333
|
+
//#endregion
|
|
334
|
+
//#region src/campaign/presets/run-eval.d.ts
|
|
335
|
+
interface RunEvalOptions<TScenario extends Scenario, TArtifact> extends Omit<RunCampaignOptions<TScenario, TArtifact>, 'runDir'> {
|
|
336
|
+
runDir: string;
|
|
337
|
+
}
|
|
338
|
+
/**
|
|
339
|
+
* Simplest evaluation preset: run scenarios through dispatch, score with judges, and return a `CampaignResult` — no optimizer, no gate, no PR.
|
|
340
|
+
*/
|
|
341
|
+
declare function runEval<TScenario extends Scenario, TArtifact>(opts: RunEvalOptions<TScenario, TArtifact>): Promise<CampaignResult<TArtifact, TScenario>>;
|
|
342
|
+
//#endregion
|
|
343
|
+
//#region src/campaign/auto-pr.d.ts
|
|
344
|
+
interface OpenAutoPrOptions<TArtifact, TScenario extends Scenario> {
|
|
345
|
+
/** Campaign result to attach to the PR. */
|
|
346
|
+
result: CampaignResult<TArtifact, TScenario>;
|
|
347
|
+
/** Gate verdict explaining the promotion. Substrate refuses to open a PR
|
|
348
|
+
* when `gate.decision !== 'ship'` — fails loud. */
|
|
349
|
+
gate: GateResult;
|
|
350
|
+
/** Promoted surface diff — typically the new system prompt addendum or
|
|
351
|
+
* full profile diff. Substrate writes it as the PR body. */
|
|
352
|
+
promotedDiff: string;
|
|
353
|
+
/** GH owner/repo target (e.g., `tangle-network/gtm-agent`). */
|
|
354
|
+
ghOwner: string;
|
|
355
|
+
ghRepo: string;
|
|
356
|
+
/** Branch name for the PR. Default `auto/<manifestHash[:12]>`. */
|
|
357
|
+
branch?: string;
|
|
358
|
+
/** PR title. Default includes manifest hash. */
|
|
359
|
+
title?: string;
|
|
360
|
+
/** Whether to actually open the PR or just dry-run. Default reads
|
|
361
|
+
* `GH_AUTO_PR_TOKEN` env — present = open, absent = dry-run. */
|
|
362
|
+
dryRun?: boolean;
|
|
363
|
+
/** Test seam — substitute `gh pr create` invocation. */
|
|
364
|
+
ghExec?: (args: string[]) => {
|
|
365
|
+
stdout: string;
|
|
366
|
+
stderr: string;
|
|
367
|
+
status: number;
|
|
368
|
+
};
|
|
369
|
+
}
|
|
370
|
+
interface OpenAutoPrResult {
|
|
371
|
+
opened: boolean;
|
|
372
|
+
prUrl?: string;
|
|
373
|
+
dryRun: boolean;
|
|
374
|
+
reason: string;
|
|
375
|
+
}
|
|
376
|
+
/**
|
|
377
|
+
* Open a GitHub PR for a gate-approved surface promotion, attaching the manifest hash, gate verdict, and diff as the PR body.
|
|
378
|
+
*/
|
|
379
|
+
declare function openAutoPr<TArtifact, TScenario extends Scenario>(options: OpenAutoPrOptions<TArtifact, TScenario>): OpenAutoPrResult;
|
|
380
|
+
//#endregion
|
|
381
|
+
//#region src/campaign/parent-selection.d.ts
|
|
382
|
+
/** Search state supplied to one parent-selection call. */
|
|
383
|
+
interface ParentSelectionContext {
|
|
384
|
+
/** Non-dominated scored surfaces across the whole run so far, including the
|
|
385
|
+
* baseline (`generation: -1`). Never empty. */
|
|
386
|
+
readonly frontier: ReadonlyArray<ParetoParent>;
|
|
387
|
+
/** Measured result of the global incumbent, the promotion bar. Under the
|
|
388
|
+
* default `selectionRankKey` the incumbent is always on `frontier`. */
|
|
389
|
+
readonly incumbent: ScoredSurfaceOutcome;
|
|
390
|
+
/** Every completed generation so far. */
|
|
391
|
+
readonly history: ReadonlyArray<GenerationRecord>;
|
|
392
|
+
/** Index of the generation about to propose. */
|
|
393
|
+
readonly generation: number;
|
|
394
|
+
}
|
|
395
|
+
/** Chooses the surface the next generation mutates. Returns one frontier
|
|
396
|
+
* parent; `runOptimization` refuses a parent it has not measured to
|
|
397
|
+
* completion or whose surface does not match its `surfaceHash`. */
|
|
398
|
+
type ParentSelector = (ctx: ParentSelectionContext) => ParetoParent;
|
|
399
|
+
interface CrowdedFrontierParentOptions {
|
|
400
|
+
/** Integer seed for the per-generation draw. The same seed, frontier, and
|
|
401
|
+
* generation index select the same parent. */
|
|
402
|
+
seed: number;
|
|
403
|
+
}
|
|
404
|
+
/**
|
|
405
|
+
* NSGA-II crowded tournament selection over the frontier. Each generation
|
|
406
|
+
* draws two distinct frontier members with a PRNG seeded from `seed` and the
|
|
407
|
+
* generation index, and keeps the one with the larger crowding distance (more
|
|
408
|
+
* isolated on the frontier). Boundary parents carry infinite distance, so a
|
|
409
|
+
* boundary parent always beats an interior one. A tie on distance falls back
|
|
410
|
+
* to the higher mean composite, then to the smaller surface hash. A frontier
|
|
411
|
+
* of one member returns that member.
|
|
412
|
+
*/
|
|
413
|
+
declare function crowdedFrontierParent(options: CrowdedFrontierParentOptions): ParentSelector;
|
|
414
|
+
//#endregion
|
|
415
|
+
//#region src/campaign/search-ledger-errors.d.ts
|
|
416
|
+
/** Base error for invalid search-ledger input or operations. */
|
|
417
|
+
declare class SearchLedgerError extends ValidationError {}
|
|
418
|
+
/** Error raised when durable search-ledger data fails an integrity check. */
|
|
419
|
+
declare class SearchLedgerIntegrityError extends SearchLedgerError {}
|
|
420
|
+
/** Error raised when an event identifier is reused with different content. */
|
|
421
|
+
declare class SearchLedgerConflictError extends SearchLedgerError {}
|
|
422
|
+
//#endregion
|
|
423
|
+
//#region src/campaign/search-ledger.d.ts
|
|
424
|
+
declare const SEARCH_LEDGER_SCHEMA: 'tangle.search-ledger.v1';
|
|
425
|
+
type SearchLedgerHash = LedgerHash;
|
|
426
|
+
type SearchSurfaceKind = 'prompt' | 'tool-contract' | 'runtime-config' | 'memory' | 'knowledge' | 'agent-profile' | 'code' | 'deployment';
|
|
427
|
+
/** Content-addressed artifact or receipt. Mutable paths are locators only; the
|
|
428
|
+
* digest and byte length bind the exact bytes used by the search. */
|
|
429
|
+
interface SearchArtifactRef {
|
|
430
|
+
role: string;
|
|
431
|
+
uri: string;
|
|
432
|
+
sha256: SearchLedgerHash;
|
|
433
|
+
byteLength: number;
|
|
434
|
+
}
|
|
435
|
+
/** Repository, dataset, or package source pinned to an immutable commit or
|
|
436
|
+
* content digest. Branches, tags, and bare package versions are rejected. */
|
|
437
|
+
interface SearchSourceRef {
|
|
438
|
+
uri: string;
|
|
439
|
+
revision: string;
|
|
440
|
+
}
|
|
441
|
+
interface SearchModelIdentity {
|
|
442
|
+
provider: string;
|
|
443
|
+
snapshot: string;
|
|
444
|
+
}
|
|
445
|
+
interface SearchCandidateSurface {
|
|
446
|
+
surfaceId: string;
|
|
447
|
+
kind: SearchSurfaceKind;
|
|
448
|
+
artifact: SearchArtifactRef;
|
|
449
|
+
}
|
|
450
|
+
interface SearchCandidateLineage {
|
|
451
|
+
/** Existing `LineageNode.id`; this ledger references rather than embeds it. */
|
|
452
|
+
lineageNodeId: string;
|
|
453
|
+
parentCandidateIds: string[];
|
|
454
|
+
generation: number;
|
|
455
|
+
proposer: string;
|
|
456
|
+
proposerSource: SearchSourceRef;
|
|
457
|
+
}
|
|
458
|
+
type SearchOperationKind = 'candidate-generation' | 'analysis' | 'selection' | 'judge' | 'other';
|
|
459
|
+
interface SearchPlannedTask {
|
|
460
|
+
taskId: string;
|
|
461
|
+
source: SearchSourceRef;
|
|
462
|
+
benchmark: SearchSourceRef;
|
|
463
|
+
/** Maximum transport attempts for this task and candidate. Only an explicit
|
|
464
|
+
* passed/failed outcome satisfies the planned denominator. */
|
|
465
|
+
maxAttempts: number;
|
|
466
|
+
}
|
|
467
|
+
interface SearchPlannedOperation {
|
|
468
|
+
operationId: string;
|
|
469
|
+
kind: SearchOperationKind;
|
|
470
|
+
}
|
|
471
|
+
interface SearchCandidateSlot {
|
|
472
|
+
slotId: string;
|
|
473
|
+
/** Planned candidate-generation call that must either produce this slot or
|
|
474
|
+
* fail before the slot can be closed. Several slots may share one batched call. */
|
|
475
|
+
generationOperationId: string;
|
|
476
|
+
}
|
|
477
|
+
interface SearchPlan {
|
|
478
|
+
/** Stable slots and their proposer calls are frozen before search begins. */
|
|
479
|
+
candidateSlots: SearchCandidateSlot[];
|
|
480
|
+
/** Every task applies to every successfully registered candidate. */
|
|
481
|
+
tasks: SearchPlannedTask[];
|
|
482
|
+
/** Non-task spend slots: proposal, analysis, selection, extra judges, etc. */
|
|
483
|
+
operations: SearchPlannedOperation[];
|
|
484
|
+
}
|
|
485
|
+
type SearchTokenAccounting = {
|
|
486
|
+
status: 'known';
|
|
487
|
+
inputTokens: number;
|
|
488
|
+
outputTokens: number;
|
|
489
|
+
cachedTokens: number;
|
|
490
|
+
} | {
|
|
491
|
+
status: 'unknown';
|
|
492
|
+
reason: string;
|
|
493
|
+
};
|
|
494
|
+
type SearchCostAccounting = {
|
|
495
|
+
status: 'known';
|
|
496
|
+
usd: number;
|
|
497
|
+
source: 'provider' | 'pricing-table' | 'free';
|
|
498
|
+
} | {
|
|
499
|
+
status: 'unknown';
|
|
500
|
+
/** Known spend may still be a lower bound when one call was unpriced. */
|
|
501
|
+
knownLowerBoundUsd: number;
|
|
502
|
+
reason: string;
|
|
503
|
+
};
|
|
504
|
+
interface SearchAttemptAccounting {
|
|
505
|
+
tokens: SearchTokenAccounting;
|
|
506
|
+
cost: SearchCostAccounting;
|
|
507
|
+
}
|
|
508
|
+
interface SearchFailureReason {
|
|
509
|
+
code: string;
|
|
510
|
+
message: string;
|
|
511
|
+
}
|
|
512
|
+
type SearchTaskOutcome = {
|
|
513
|
+
status: 'passed';
|
|
514
|
+
score: number;
|
|
515
|
+
metrics: Record<string, number>;
|
|
516
|
+
} | {
|
|
517
|
+
status: 'failed';
|
|
518
|
+
score: number;
|
|
519
|
+
metrics: Record<string, number>;
|
|
520
|
+
failure: SearchFailureReason;
|
|
521
|
+
} | {
|
|
522
|
+
status: 'errored';
|
|
523
|
+
metrics: Record<string, number>;
|
|
524
|
+
error: SearchFailureReason & {
|
|
525
|
+
retryable: boolean;
|
|
526
|
+
};
|
|
527
|
+
};
|
|
528
|
+
type SearchSurfaceEffect = {
|
|
529
|
+
status: 'measured';
|
|
530
|
+
metric: string;
|
|
531
|
+
baselineValue: number;
|
|
532
|
+
candidateValue: number;
|
|
533
|
+
delta: number;
|
|
534
|
+
} | {
|
|
535
|
+
status: 'not-measured';
|
|
536
|
+
reason: string;
|
|
537
|
+
};
|
|
538
|
+
/** Per-attempt proof that a declared candidate surface was or was not active,
|
|
539
|
+
* plus measured effect when the experiment supports attribution. */
|
|
540
|
+
interface SearchSurfaceEvidence {
|
|
541
|
+
surfaceId: string;
|
|
542
|
+
fired: boolean;
|
|
543
|
+
firingCount: number;
|
|
544
|
+
effect: SearchSurfaceEffect;
|
|
545
|
+
evidence: SearchArtifactRef[];
|
|
546
|
+
}
|
|
547
|
+
interface SearchLedgerEventBase {
|
|
548
|
+
eventId: string;
|
|
549
|
+
occurredAt: string;
|
|
550
|
+
artifacts: SearchArtifactRef[];
|
|
551
|
+
}
|
|
552
|
+
interface SearchPlannedEvent extends SearchLedgerEventBase {
|
|
553
|
+
kind: 'search-planned';
|
|
554
|
+
plan: SearchPlan;
|
|
555
|
+
}
|
|
556
|
+
/** Additional candidate slots and operations for a search whose length is not
|
|
557
|
+
* known when it starts. The plan stays the first event and the planned task
|
|
558
|
+
* denominator stays frozen: extending tasks would retroactively reopen
|
|
559
|
+
* candidates that already closed theirs. */
|
|
560
|
+
interface SearchPlanExtendedEvent extends SearchLedgerEventBase {
|
|
561
|
+
kind: 'search-plan-extended';
|
|
562
|
+
extension: {
|
|
563
|
+
candidateSlots: SearchCandidateSlot[];
|
|
564
|
+
operations: SearchPlannedOperation[];
|
|
565
|
+
};
|
|
566
|
+
}
|
|
567
|
+
interface SearchCandidateRegisteredEvent extends SearchLedgerEventBase {
|
|
568
|
+
kind: 'candidate-registered';
|
|
569
|
+
slotId: string;
|
|
570
|
+
generationOperationId: string;
|
|
571
|
+
candidateId: string;
|
|
572
|
+
lineage: SearchCandidateLineage;
|
|
573
|
+
surfaces: SearchCandidateSurface[];
|
|
574
|
+
}
|
|
575
|
+
interface SearchCandidateSlotClosedEvent extends SearchLedgerEventBase {
|
|
576
|
+
kind: 'candidate-slot-closed';
|
|
577
|
+
slotId: string;
|
|
578
|
+
generationOperationId: string;
|
|
579
|
+
reason: SearchFailureReason;
|
|
580
|
+
}
|
|
581
|
+
interface SearchTaskAttemptedEvent extends SearchLedgerEventBase {
|
|
582
|
+
kind: 'task-attempted';
|
|
583
|
+
candidateId: string;
|
|
584
|
+
runId: string;
|
|
585
|
+
attemptIndex: number;
|
|
586
|
+
task: {
|
|
587
|
+
taskId: string;
|
|
588
|
+
source: SearchSourceRef;
|
|
589
|
+
};
|
|
590
|
+
identity: {
|
|
591
|
+
model: SearchModelIdentity;
|
|
592
|
+
agent: SearchSourceRef;
|
|
593
|
+
benchmark: SearchSourceRef;
|
|
594
|
+
};
|
|
595
|
+
outcome: SearchTaskOutcome;
|
|
596
|
+
accounting: SearchAttemptAccounting;
|
|
597
|
+
surfaceEvidence: SearchSurfaceEvidence[];
|
|
598
|
+
}
|
|
599
|
+
interface SearchOperationRecordedEvent extends SearchLedgerEventBase {
|
|
600
|
+
kind: 'search-operation-recorded';
|
|
601
|
+
operationId: string;
|
|
602
|
+
operationKind: SearchOperationKind;
|
|
603
|
+
execution: {
|
|
604
|
+
kind: 'model';
|
|
605
|
+
model: SearchModelIdentity;
|
|
606
|
+
source: SearchSourceRef;
|
|
607
|
+
} | {
|
|
608
|
+
kind: 'deterministic';
|
|
609
|
+
source: SearchSourceRef;
|
|
610
|
+
};
|
|
611
|
+
outcome: {
|
|
612
|
+
status: 'completed';
|
|
613
|
+
} | {
|
|
614
|
+
status: 'partial';
|
|
615
|
+
failure: SearchFailureReason;
|
|
616
|
+
} | {
|
|
617
|
+
status: 'failed';
|
|
618
|
+
failure: SearchFailureReason;
|
|
619
|
+
};
|
|
620
|
+
accounting: SearchAttemptAccounting;
|
|
621
|
+
}
|
|
622
|
+
interface SearchCandidateDecidedEvent extends SearchLedgerEventBase {
|
|
623
|
+
kind: 'candidate-decided';
|
|
624
|
+
candidateId: string;
|
|
625
|
+
decision: {
|
|
626
|
+
status: 'selected';
|
|
627
|
+
} | {
|
|
628
|
+
status: 'rejected';
|
|
629
|
+
reason: SearchFailureReason;
|
|
630
|
+
};
|
|
631
|
+
}
|
|
632
|
+
interface SearchCompletedEvent extends SearchLedgerEventBase {
|
|
633
|
+
kind: 'search-completed';
|
|
634
|
+
result: {
|
|
635
|
+
status: 'selected';
|
|
636
|
+
candidateId: string;
|
|
637
|
+
} | {
|
|
638
|
+
status: 'all-rejected';
|
|
639
|
+
reason: SearchFailureReason;
|
|
640
|
+
};
|
|
641
|
+
}
|
|
642
|
+
type SearchLedgerEvent = SearchPlannedEvent | SearchPlanExtendedEvent | SearchCandidateRegisteredEvent | SearchCandidateSlotClosedEvent | SearchTaskAttemptedEvent | SearchOperationRecordedEvent | SearchCandidateDecidedEvent | SearchCompletedEvent;
|
|
643
|
+
interface SearchLedgerEntry {
|
|
644
|
+
schema: typeof SEARCH_LEDGER_SCHEMA;
|
|
645
|
+
campaignId: string;
|
|
646
|
+
sequence: number;
|
|
647
|
+
previousHash: SearchLedgerHash | null;
|
|
648
|
+
event: SearchLedgerEvent;
|
|
649
|
+
entryHash: SearchLedgerHash;
|
|
650
|
+
}
|
|
651
|
+
type SearchAccountingAudit = {
|
|
652
|
+
status: 'known';
|
|
653
|
+
inputTokens: number;
|
|
654
|
+
outputTokens: number;
|
|
655
|
+
cachedTokens: number;
|
|
656
|
+
costUsd: number;
|
|
657
|
+
} | {
|
|
658
|
+
status: 'partial';
|
|
659
|
+
knownInputTokens: number;
|
|
660
|
+
knownOutputTokens: number;
|
|
661
|
+
knownCachedTokens: number;
|
|
662
|
+
knownCostUsd: number;
|
|
663
|
+
unknownTokenEventIds: string[];
|
|
664
|
+
unknownCostEventIds: string[];
|
|
665
|
+
};
|
|
666
|
+
interface SearchLedgerAudit {
|
|
667
|
+
campaignId: string;
|
|
668
|
+
eventCount: number;
|
|
669
|
+
candidateCount: number;
|
|
670
|
+
closedCandidateSlotCount: number;
|
|
671
|
+
attemptCount: number;
|
|
672
|
+
operationCount: number;
|
|
673
|
+
outcomes: {
|
|
674
|
+
passed: number;
|
|
675
|
+
failed: number;
|
|
676
|
+
errored: number;
|
|
677
|
+
};
|
|
678
|
+
operationOutcomes: {
|
|
679
|
+
completed: number;
|
|
680
|
+
partial: number;
|
|
681
|
+
failed: number;
|
|
682
|
+
};
|
|
683
|
+
decisions: {
|
|
684
|
+
selected: number;
|
|
685
|
+
rejected: number;
|
|
686
|
+
pending: number;
|
|
687
|
+
};
|
|
688
|
+
expected: {
|
|
689
|
+
candidateSlots: number;
|
|
690
|
+
taskOutcomes: number;
|
|
691
|
+
operations: number;
|
|
692
|
+
missingCandidateSlots: string[];
|
|
693
|
+
missingTaskOutcomes: string[];
|
|
694
|
+
missingOperations: string[];
|
|
695
|
+
};
|
|
696
|
+
status: 'in-progress' | 'selected' | 'all-rejected';
|
|
697
|
+
selectedCandidateId: string | null;
|
|
698
|
+
accounting: SearchAccountingAudit;
|
|
699
|
+
headHash: SearchLedgerHash | null;
|
|
700
|
+
}
|
|
701
|
+
interface SearchLedgerReplay {
|
|
702
|
+
entries: SearchLedgerEntry[];
|
|
703
|
+
plan: SearchPlannedEvent | null;
|
|
704
|
+
/** Appended plan extensions, in ledger order. The effective plan is the
|
|
705
|
+
* first plan event merged with these; `audit.expected` counts the merge. */
|
|
706
|
+
planExtensions: SearchPlanExtendedEvent[];
|
|
707
|
+
candidates: SearchCandidateRegisteredEvent[];
|
|
708
|
+
closedCandidateSlots: SearchCandidateSlotClosedEvent[];
|
|
709
|
+
attempts: SearchTaskAttemptedEvent[];
|
|
710
|
+
operations: SearchOperationRecordedEvent[];
|
|
711
|
+
decisions: SearchCandidateDecidedEvent[];
|
|
712
|
+
completion: SearchCompletedEvent | null;
|
|
713
|
+
audit: SearchLedgerAudit;
|
|
714
|
+
}
|
|
715
|
+
interface SearchLedgerAppendResult {
|
|
716
|
+
entry: SearchLedgerEntry;
|
|
717
|
+
/** False when the exact event was already durably present. */
|
|
718
|
+
appended: boolean;
|
|
719
|
+
replay: SearchLedgerReplay;
|
|
720
|
+
}
|
|
721
|
+
/** Validate and return a canonical copy. Arrays whose order is not semantic are
|
|
722
|
+
* sorted so retries from different processes produce byte-identical events. */
|
|
723
|
+
declare function validateSearchLedgerEvent(input: unknown): SearchLedgerEvent;
|
|
724
|
+
/**
|
|
725
|
+
* How this ledger uses its trusted head — the `(sequence, entryHash)` pin kept
|
|
726
|
+
* in the sibling `<path>.head` file that a hash chain needs to prove entries
|
|
727
|
+
* were not deleted from the end. `ledger-core/trusted-head.ts` holds the threat
|
|
728
|
+
* model.
|
|
729
|
+
*
|
|
730
|
+
* - `pin` (default): every append records the new head, and a pin that is
|
|
731
|
+
* present is verified on every read.
|
|
732
|
+
* - `require`: additionally refuses to read a non-empty ledger whose pin is
|
|
733
|
+
* gone, so deleting the sibling file cannot downgrade the guarantee. Only for
|
|
734
|
+
* ledgers written under `pin` from their first entry.
|
|
735
|
+
* - `off`: chain verification only. Truncation to a valid shorter prefix is
|
|
736
|
+
* undetectable.
|
|
737
|
+
*/
|
|
738
|
+
type SearchLedgerTrustedHeadMode = 'pin' | 'require' | 'off';
|
|
739
|
+
interface OpenSearchLedgerOptions {
|
|
740
|
+
path: string;
|
|
741
|
+
campaignId: string;
|
|
742
|
+
trustedHead?: SearchLedgerTrustedHeadMode;
|
|
743
|
+
}
|
|
744
|
+
interface SearchLedger {
|
|
745
|
+
readonly path: string;
|
|
746
|
+
readonly campaignId: string;
|
|
747
|
+
/** Sibling file holding this ledger's trusted head. */
|
|
748
|
+
readonly trustedHeadPath: string;
|
|
749
|
+
append(event: SearchLedgerEvent): Promise<SearchLedgerAppendResult>;
|
|
750
|
+
replay(): Promise<SearchLedgerReplay>;
|
|
751
|
+
/** The pinned head, or null when this ledger has never been pinned. */
|
|
752
|
+
trustedHead(): Promise<LedgerTrustedHead | null>;
|
|
753
|
+
/** Pin the current verified head: how a ledger written under `off`, or one
|
|
754
|
+
* whose pin file was removed, acquires a pin without rewriting a byte. */
|
|
755
|
+
pinTrustedHead(): Promise<LedgerTrustedHead>;
|
|
756
|
+
/** Discard this ledger's pin, reporting what was discarded. Deleting or
|
|
757
|
+
* rebuilding the ledger file leaves a pin naming history the file no longer
|
|
758
|
+
* carries, and every later read is refused because that is exactly the
|
|
759
|
+
* deletion the pin exists to catch; clearing is the supported way to abandon
|
|
760
|
+
* that history on purpose. It gives up the deletion guarantee for every entry
|
|
761
|
+
* the pin covered. */
|
|
762
|
+
clearTrustedHead(): Promise<LedgerTrustedHeadRemoval>;
|
|
763
|
+
}
|
|
764
|
+
/** Open a durable filesystem search ledger. Construction performs no I/O; the
|
|
765
|
+
* first `append` or `replay` validates the complete existing file. */
|
|
766
|
+
declare function openSearchLedger(options: OpenSearchLedgerOptions): SearchLedger;
|
|
767
|
+
/** Append-only file-backed search ledger with idempotent writes and replay. */
|
|
768
|
+
declare class FileSearchLedger implements SearchLedger {
|
|
769
|
+
readonly path: string;
|
|
770
|
+
readonly campaignId: string;
|
|
771
|
+
readonly trustedHeadPath: string;
|
|
772
|
+
private readonly trustedHeadMode;
|
|
773
|
+
private readonly journal;
|
|
774
|
+
constructor(path: string, campaignId: string, trustedHead?: SearchLedgerTrustedHeadMode);
|
|
775
|
+
replay(): Promise<SearchLedgerReplay>;
|
|
776
|
+
append(input: SearchLedgerEvent): Promise<SearchLedgerAppendResult>;
|
|
777
|
+
trustedHead(): Promise<LedgerTrustedHead | null>;
|
|
778
|
+
pinTrustedHead(): Promise<LedgerTrustedHead>;
|
|
779
|
+
clearTrustedHead(): Promise<LedgerTrustedHeadRemoval>;
|
|
780
|
+
}
|
|
781
|
+
//#endregion
|
|
782
|
+
//#region src/campaign/search-history-receipt.d.ts
|
|
783
|
+
declare const SEARCH_HISTORY_RECEIPT_SCHEMA_VERSION: '1.0.0';
|
|
784
|
+
declare const SEARCH_HISTORY_RECEIPT_DIGEST_ALGORITHM: 'rfc8785-sha256';
|
|
785
|
+
/** Bounded projection of the canonical replay audit. Exact ids stay in SearchLedger. */
|
|
786
|
+
interface SearchHistoryAuditSummary {
|
|
787
|
+
readonly campaignId: string;
|
|
788
|
+
readonly headHash: SearchLedgerHash | null;
|
|
789
|
+
readonly status: SearchLedgerAudit['status'];
|
|
790
|
+
readonly selectedCandidateId: string | null;
|
|
791
|
+
readonly eventCount: number;
|
|
792
|
+
readonly candidateCount: number;
|
|
793
|
+
readonly closedCandidateSlotCount: number;
|
|
794
|
+
readonly attemptCount: number;
|
|
795
|
+
readonly operationCount: number;
|
|
796
|
+
readonly expectedCandidateSlots: number;
|
|
797
|
+
readonly expectedTaskOutcomes: number;
|
|
798
|
+
readonly expectedOperations: number;
|
|
799
|
+
readonly missingCandidateSlots: number;
|
|
800
|
+
readonly missingTaskOutcomes: number;
|
|
801
|
+
readonly missingOperations: number;
|
|
802
|
+
readonly pendingDecisions: number;
|
|
803
|
+
readonly hasPlan: boolean;
|
|
804
|
+
readonly hasCompletion: boolean;
|
|
805
|
+
}
|
|
806
|
+
/**
|
|
807
|
+
* A bounded proof envelope over one canonical SearchLedger replay.
|
|
808
|
+
*
|
|
809
|
+
* The content-addressed ledger remains the sole rich history. This receipt binds
|
|
810
|
+
* its producer/run identity, exact audit digest, bounded completeness summary,
|
|
811
|
+
* and its own canonical digest. Consumers needing candidate ids, attempts,
|
|
812
|
+
* failures, accounting gaps, or decisions read and replay the canonical ledger.
|
|
813
|
+
*/
|
|
814
|
+
interface SearchHistoryReceipt {
|
|
815
|
+
readonly schemaVersion: typeof SEARCH_HISTORY_RECEIPT_SCHEMA_VERSION;
|
|
816
|
+
readonly kind: 'search-history-receipt';
|
|
817
|
+
readonly digestAlgorithm: typeof SEARCH_HISTORY_RECEIPT_DIGEST_ALGORITHM;
|
|
818
|
+
readonly receiptDigest: SearchLedgerHash;
|
|
819
|
+
/** Stable producer identity, for example an OptimizationMethod name. */
|
|
820
|
+
readonly producerId: string;
|
|
821
|
+
/** Concrete optimizer/runtime invocation that produced the ledger. */
|
|
822
|
+
readonly runId: string;
|
|
823
|
+
readonly ledger: SearchArtifactRef;
|
|
824
|
+
/** Digest of the complete SearchLedgerAudit produced by canonical replay. */
|
|
825
|
+
readonly auditDigest: SearchLedgerHash;
|
|
826
|
+
readonly summary: SearchHistoryAuditSummary;
|
|
827
|
+
readonly complete: boolean;
|
|
828
|
+
readonly incompleteReasons: readonly string[];
|
|
829
|
+
}
|
|
830
|
+
interface CreateSearchHistoryReceiptInput {
|
|
831
|
+
readonly producerId: string;
|
|
832
|
+
readonly runId: string;
|
|
833
|
+
/** Content-addressed canonical SearchLedger JSONL artifact. */
|
|
834
|
+
readonly ledger: SearchArtifactRef;
|
|
835
|
+
/** The result returned by SearchLedger.replay(). */
|
|
836
|
+
readonly replay: SearchLedgerReplay;
|
|
837
|
+
}
|
|
838
|
+
type SearchHistoryPolicy = 'allow-missing' | 'require-complete';
|
|
839
|
+
interface SearchHistoryCoverageRow {
|
|
840
|
+
readonly producerId: string;
|
|
841
|
+
readonly status: 'complete' | 'incomplete' | 'missing';
|
|
842
|
+
readonly reasons: readonly string[];
|
|
843
|
+
readonly receipt?: SearchHistoryReceipt;
|
|
844
|
+
}
|
|
845
|
+
interface SearchHistoryCoverage {
|
|
846
|
+
readonly policy: SearchHistoryPolicy;
|
|
847
|
+
readonly allComplete: boolean;
|
|
848
|
+
readonly producers: readonly SearchHistoryCoverageRow[];
|
|
849
|
+
}
|
|
850
|
+
declare class SearchHistoryRequiredError extends Error {
|
|
851
|
+
readonly producerId: string;
|
|
852
|
+
readonly reasons: readonly string[];
|
|
853
|
+
constructor(producerId: string, reasons: readonly string[]);
|
|
854
|
+
}
|
|
855
|
+
/** Build a bounded receipt from the projection returned by canonical ledger replay. */
|
|
856
|
+
declare function createSearchHistoryReceipt(input: CreateSearchHistoryReceiptInput): SearchHistoryReceipt;
|
|
857
|
+
/** Verify the bounded receipt. Full ledger bytes are verified by SearchLedger. */
|
|
858
|
+
declare function verifySearchHistoryReceipt(receipt: SearchHistoryReceipt): SearchHistoryReceipt;
|
|
859
|
+
/** Prove that a receipt still describes the exact canonical replay supplied. */
|
|
860
|
+
declare function assertSearchHistoryMatchesReplay(receipt: SearchHistoryReceipt, replay: SearchLedgerReplay): void;
|
|
861
|
+
/** Require a receipt owned by this producer and a terminal, denominator-complete history. */
|
|
862
|
+
declare function assertCompleteSearchHistory(producerId: string, receipt: SearchHistoryReceipt | undefined): asserts receipt is SearchHistoryReceipt;
|
|
863
|
+
/** Classify one producer's history without treating malformed evidence as absence. */
|
|
864
|
+
declare function searchHistoryCoverageRow(producerId: string, receipt: SearchHistoryReceipt | undefined): SearchHistoryCoverageRow;
|
|
865
|
+
//#endregion
|
|
866
|
+
//#region src/campaign/gepa-candidate-population.d.ts
|
|
867
|
+
interface GepaCandidatePopulationSummary {
|
|
868
|
+
readonly scope: 'gepa-candidate-population';
|
|
869
|
+
readonly path: string;
|
|
870
|
+
readonly sha256: `sha256:${string}`;
|
|
871
|
+
readonly bytes: number;
|
|
872
|
+
readonly runId: string;
|
|
873
|
+
readonly candidates: number;
|
|
874
|
+
readonly bestIndex: number;
|
|
875
|
+
readonly maxCandidates: number;
|
|
876
|
+
readonly maxCandidateChars: number;
|
|
877
|
+
readonly scenarioIds: readonly string[];
|
|
878
|
+
readonly surfaceKind: 'text' | 'components';
|
|
879
|
+
}
|
|
880
|
+
interface GepaCandidateSelectionScore {
|
|
881
|
+
readonly scenarioId: string;
|
|
882
|
+
readonly score: number;
|
|
883
|
+
}
|
|
884
|
+
interface GepaCandidatePopulationCandidate {
|
|
885
|
+
/** Zero-based index assigned by the exact GEPA result. */
|
|
886
|
+
readonly index: number;
|
|
887
|
+
readonly candidate: ExternalTextCandidate;
|
|
888
|
+
readonly candidateHash: string;
|
|
889
|
+
readonly candidateDigest: `sha256:${string}`;
|
|
890
|
+
/** Exact GEPA parent indices. The seed candidate has one null parent. */
|
|
891
|
+
readonly parentIndices: readonly (number | null)[];
|
|
892
|
+
/** Null means GEPA had no selection score for this candidate. */
|
|
893
|
+
readonly aggregateScore: number | null;
|
|
894
|
+
readonly selectionScores: readonly GepaCandidateSelectionScore[];
|
|
895
|
+
readonly discoveryEvaluationCount: number;
|
|
896
|
+
}
|
|
897
|
+
interface GepaCandidatePopulationArtifact {
|
|
898
|
+
readonly summary: GepaCandidatePopulationSummary;
|
|
899
|
+
readonly runId: string;
|
|
900
|
+
readonly bestIndex: number;
|
|
901
|
+
readonly candidates: readonly GepaCandidatePopulationCandidate[];
|
|
902
|
+
}
|
|
903
|
+
/**
|
|
904
|
+
* Read GEPA's exact candidate graph from the artifact addressed by method provenance.
|
|
905
|
+
*
|
|
906
|
+
* The reader checks the supplied digest, declared byte count, run identity,
|
|
907
|
+
* candidate surfaces, parent graph, selection scores, and configured bounds.
|
|
908
|
+
* This proves that the bytes match the supplied summary. The caller remains
|
|
909
|
+
* responsible for obtaining that summary from trusted method provenance.
|
|
910
|
+
*/
|
|
911
|
+
declare function readGepaCandidatePopulationArtifact(input: {
|
|
912
|
+
summary: GepaCandidatePopulationSummary;
|
|
913
|
+
storage?: CampaignStorage;
|
|
914
|
+
}): GepaCandidatePopulationArtifact;
|
|
915
|
+
//#endregion
|
|
916
|
+
//#region src/campaign/search-ledger-recording.d.ts
|
|
917
|
+
/** How a search operation executed. The shape the ledger event records. */
|
|
918
|
+
type SearchExecutionIdentity = SearchOperationRecordedEvent['execution'];
|
|
919
|
+
/** Immutable identities the ledger requires and a campaign cannot infer. */
|
|
920
|
+
interface SearchRunIdentity {
|
|
921
|
+
/** The agent implementation under optimization. */
|
|
922
|
+
agent: SearchSourceRef;
|
|
923
|
+
/** The candidate generator: a model call or deterministic code. */
|
|
924
|
+
proposer: SearchExecutionIdentity;
|
|
925
|
+
/** The code that plans the search and selects its winner. */
|
|
926
|
+
search: SearchSourceRef;
|
|
927
|
+
/** Model the agent runs. Used only for a cell that reported none. */
|
|
928
|
+
model: SearchModelIdentity;
|
|
929
|
+
}
|
|
930
|
+
interface SearchLedgerBinding {
|
|
931
|
+
ledger: SearchLedger;
|
|
932
|
+
identity: SearchRunIdentity;
|
|
933
|
+
}
|
|
934
|
+
/** One proposed candidate, before it is measured. */
|
|
935
|
+
interface ProposedSearchCandidate {
|
|
936
|
+
surface: MutableSurface;
|
|
937
|
+
surfaceHash: string;
|
|
938
|
+
label?: string;
|
|
939
|
+
}
|
|
940
|
+
/** One measured candidate, after its campaign scored. */
|
|
941
|
+
interface MeasuredSearchCandidate<TArtifact> {
|
|
942
|
+
surface: MutableSurface;
|
|
943
|
+
surfaceHash: string;
|
|
944
|
+
cells: ReadonlyArray<CampaignCellResult<TArtifact>>;
|
|
945
|
+
runDir: string;
|
|
946
|
+
/** False when the candidate missed a designed cell. */
|
|
947
|
+
coverageComplete: boolean;
|
|
948
|
+
}
|
|
949
|
+
interface SearchRecorderOptions<TScenario extends Scenario> {
|
|
950
|
+
binding: SearchLedgerBinding;
|
|
951
|
+
storage: CampaignStorage;
|
|
952
|
+
runDir: string;
|
|
953
|
+
scenarios: ReadonlyArray<TScenario>;
|
|
954
|
+
reps: number;
|
|
955
|
+
maxGenerations: number;
|
|
956
|
+
populationSize: number;
|
|
957
|
+
/** Identity of the exact campaign design; the task benchmark pin. */
|
|
958
|
+
splitDigest: `sha256:${string}`;
|
|
959
|
+
/** Proposer label recorded on every candidate lineage. */
|
|
960
|
+
proposerLabel: string;
|
|
961
|
+
costLedger: CostLedgerHandle;
|
|
962
|
+
}
|
|
963
|
+
/**
|
|
964
|
+
* Recorder for one `runOptimization` run: `open()`, then `recordGeneration()`
|
|
965
|
+
* and `recordResults()` per generation, then `finish()`.
|
|
966
|
+
*
|
|
967
|
+
* Every event id is derived from the run, and an id already durable is not
|
|
968
|
+
* appended again, so a resumed run continues one ledger instead of conflicting
|
|
969
|
+
* with its own history.
|
|
970
|
+
*/
|
|
971
|
+
declare class SearchRecorder<TScenario extends Scenario, TArtifact> {
|
|
972
|
+
private readonly opts;
|
|
973
|
+
private readonly tasks;
|
|
974
|
+
private readonly registered;
|
|
975
|
+
private readonly order;
|
|
976
|
+
private readonly coverage;
|
|
977
|
+
private readonly openSlots;
|
|
978
|
+
private readonly durableEventIds;
|
|
979
|
+
private lastStampMs;
|
|
980
|
+
private proposalReceiptCount;
|
|
981
|
+
private constructor();
|
|
982
|
+
/** Open the recorder and append the plan. An existing ledger for the same
|
|
983
|
+
* run is re-read first, so a resumed run keeps one plan and one lineage. */
|
|
984
|
+
static open<TScenario extends Scenario, TArtifact>(opts: SearchRecorderOptions<TScenario>): Promise<SearchRecorder<TScenario, TArtifact>>;
|
|
985
|
+
/**
|
|
986
|
+
* Record one generation's candidate-generation call and the candidates it
|
|
987
|
+
* produced. A proposal larger than the planned population extends the plan
|
|
988
|
+
* with the extra slots; a proposal that fills fewer closes the rest.
|
|
989
|
+
*/
|
|
990
|
+
recordGeneration(input: {
|
|
991
|
+
generation: number;
|
|
992
|
+
parentSurfaceHash: string;
|
|
993
|
+
candidates: ReadonlyArray<ProposedSearchCandidate>;
|
|
994
|
+
}): Promise<void>;
|
|
995
|
+
/** Append one task attempt per designed cell of each candidate campaign. */
|
|
996
|
+
recordResults(candidates: ReadonlyArray<MeasuredSearchCandidate<TArtifact>>): Promise<void>;
|
|
997
|
+
/**
|
|
998
|
+
* Close the search: unreached generations, the selection operation, one
|
|
999
|
+
* decision per candidate, then the terminal event.
|
|
1000
|
+
*
|
|
1001
|
+
* The terminal event is appended only when canonical replay accounts for the
|
|
1002
|
+
* whole planned denominator. An interrupted or partly unscored search stays
|
|
1003
|
+
* `in-progress` and its receipt reports the exact gap, instead of claiming a
|
|
1004
|
+
* closed search.
|
|
1005
|
+
*/
|
|
1006
|
+
finish(input: {
|
|
1007
|
+
winnerSurfaceHash: string;
|
|
1008
|
+
generationsRun: number;
|
|
1009
|
+
runId: string;
|
|
1010
|
+
}): Promise<SearchHistoryReceipt>;
|
|
1011
|
+
/** Bounded receipt over the exact ledger bytes this run produced. */
|
|
1012
|
+
receipt(runId: string): Promise<SearchHistoryReceipt>;
|
|
1013
|
+
/** Read an existing ledger for this run so a resume continues it. */
|
|
1014
|
+
private hydrate;
|
|
1015
|
+
private plan;
|
|
1016
|
+
private registeredSlot;
|
|
1017
|
+
private remember;
|
|
1018
|
+
private recordGenerationOperation;
|
|
1019
|
+
private closeSlot;
|
|
1020
|
+
/** Spend booked to candidate generation since the previous generation. */
|
|
1021
|
+
private proposalAccounting;
|
|
1022
|
+
private cellModel;
|
|
1023
|
+
private proposalArtifact;
|
|
1024
|
+
/** Write one canonical evidence document and return its content address. */
|
|
1025
|
+
private writeArtifact;
|
|
1026
|
+
private append;
|
|
1027
|
+
/** Non-decreasing ISO stamps; the ledger refuses an event that moves back. */
|
|
1028
|
+
private stamp;
|
|
1029
|
+
}
|
|
1030
|
+
/**
|
|
1031
|
+
* Record an optimizer's own candidate graph into the same ledger.
|
|
1032
|
+
*
|
|
1033
|
+
* A complete optimization method searches inside its own process and reports
|
|
1034
|
+
* one artifact when it finishes: the candidate population, with each
|
|
1035
|
+
* candidate's parents and its score per selection scenario. This turns that
|
|
1036
|
+
* artifact into the canonical event stream, so a first-party method returns
|
|
1037
|
+
* the same `SearchHistoryReceipt` the in-process loop returns, and
|
|
1038
|
+
* `compareOptimizationMethods({ searchHistoryPolicy: 'require-complete' })`
|
|
1039
|
+
* accepts it.
|
|
1040
|
+
*
|
|
1041
|
+
* A candidate the optimizer left unscored on a planned scenario leaves the
|
|
1042
|
+
* planned denominator open, so the receipt reports the gap instead of closing
|
|
1043
|
+
* the search.
|
|
1044
|
+
*/
|
|
1045
|
+
declare function recordCandidatePopulationSearch<TScenario extends Scenario>(input: {
|
|
1046
|
+
ledger: SearchLedger;
|
|
1047
|
+
storage: CampaignStorage;
|
|
1048
|
+
runDir: string;
|
|
1049
|
+
identity: SearchRunIdentity;
|
|
1050
|
+
population: GepaCandidatePopulationArtifact;
|
|
1051
|
+
/** Scenarios the optimizer selected on. Must cover the population's ids. */
|
|
1052
|
+
scenarios: ReadonlyArray<TScenario>;
|
|
1053
|
+
/** Spend the optimizer booked to its own candidate generation. */
|
|
1054
|
+
generationAccounting: SearchAttemptAccounting;
|
|
1055
|
+
producerId: string;
|
|
1056
|
+
runId: string;
|
|
1057
|
+
}): Promise<SearchHistoryReceipt>;
|
|
1058
|
+
//#endregion
|
|
1059
|
+
//#region src/campaign/presets/run-optimization.d.ts
|
|
1060
|
+
interface PremeasuredOptimizationBaseline<TArtifact, TScenario extends Scenario> {
|
|
1061
|
+
/** Hash of the exact surface that produced `campaign`. */
|
|
1062
|
+
surfaceHash: string;
|
|
1063
|
+
/** Complete prior measurement reused by identity, including artifactsByPath. */
|
|
1064
|
+
campaign: CampaignResult<TArtifact, TScenario>;
|
|
1065
|
+
}
|
|
1066
|
+
interface RunOptimizationBaseOptions<TScenario extends Scenario, TArtifact> extends Omit<RunCampaignOptions<TScenario, TArtifact>, 'dispatch'> {
|
|
1067
|
+
/** Initial mutable surface (typically system prompt or addendum). */
|
|
1068
|
+
baselineSurface: MutableSurface;
|
|
1069
|
+
/**
|
|
1070
|
+
* Complete prior measurement of `baselineSurface`. When present,
|
|
1071
|
+
* `runOptimization` validates its surface, scenario split, seed, reps, and
|
|
1072
|
+
* normal campaign coverage, then skips the baseline campaign entirely — no
|
|
1073
|
+
* dispatch or resumability-cache lookup. Candidate campaigns still run
|
|
1074
|
+
* normally. Prior spend remains in the imported campaign aggregates and is
|
|
1075
|
+
* not added again to this continuation's CostLedger.
|
|
1076
|
+
*/
|
|
1077
|
+
premeasuredBaseline?: PremeasuredOptimizationBaseline<TArtifact, TScenario>;
|
|
1078
|
+
/** Dispatcher that takes the CURRENT surface + scenario → artifact. */
|
|
1079
|
+
dispatchWithSurface: (surface: MutableSurface, scenario: TScenario, ctx: Parameters<RunCampaignOptions<TScenario, TArtifact>['dispatch']>[1]) => Promise<TArtifact>;
|
|
1080
|
+
/** The candidate-generation strategy. */
|
|
1081
|
+
proposer: SurfaceProposer<ProposalFinding>;
|
|
1082
|
+
populationSize: number;
|
|
1083
|
+
maxGenerations: number;
|
|
1084
|
+
/** Candidate campaigns run at once. Default 1. Total concurrent cells are
|
|
1085
|
+
* bounded by candidateConcurrency * maxConcurrency. */
|
|
1086
|
+
candidateConcurrency?: number;
|
|
1087
|
+
/** DEPTH knob forwarded to the proposer's `propose()` — max iterations the
|
|
1088
|
+
* agentic generator may take per candidate. */
|
|
1089
|
+
maxImprovementShots?: number;
|
|
1090
|
+
/** Search or observed-production findings forwarded to candidate generation. */
|
|
1091
|
+
findings?: ReadonlyArray<ProposalFinding>;
|
|
1092
|
+
/** Per-generation findings producer. Runs once on the BASELINE campaign
|
|
1093
|
+
* (as `generation: -1`, the baseline convention) before generation 0
|
|
1094
|
+
* proposes — so even a single-generation run proposes with trace context —
|
|
1095
|
+
* and then after each generation's candidates are scored with that
|
|
1096
|
+
* generation's results; whatever it returns REPLACES `ctx.findings` for the
|
|
1097
|
+
* NEXT `propose()`, so the diagnosis is refreshed each round instead
|
|
1098
|
+
* of being a static one-shot. Generic by design: the substrate does not
|
|
1099
|
+
* import an analyst — the consumer plugs its trace-analyst registry / HALO
|
|
1100
|
+
* here (reading the per-candidate `runDir` traces). When absent, findings
|
|
1101
|
+
* stay the static `opts.findings`. */
|
|
1102
|
+
analyzeGeneration?: (input: {
|
|
1103
|
+
generation: number;
|
|
1104
|
+
runDir: string;
|
|
1105
|
+
candidates: Array<{
|
|
1106
|
+
surfaceHash: string;
|
|
1107
|
+
campaign: CampaignResult<TArtifact, TScenario>;
|
|
1108
|
+
composite: number | null;
|
|
1109
|
+
}>;
|
|
1110
|
+
history: GenerationRecord[];
|
|
1111
|
+
/** Shared run spend account and receipt attribution phase. */
|
|
1112
|
+
costLedger?: CostLedgerHandle;
|
|
1113
|
+
costPhase?: string;
|
|
1114
|
+
}) => Promise<ReadonlyArray<ProposalFinding>>;
|
|
1115
|
+
/**
|
|
1116
|
+
* Optional override for how the WINNER is selected among coverage-complete
|
|
1117
|
+
* candidates (and how the incumbent bar is set). Returns a lexicographic rank
|
|
1118
|
+
* key — each element higher-is-better; candidates are ranked by descending key
|
|
1119
|
+
* (`compareRankKeys`) and the top must STRICTLY beat the incumbent's key to
|
|
1120
|
+
* promote. Defaults to `[campaignMeanComposite(campaign)]`, i.e. the historical
|
|
1121
|
+
* scalar-mean ranking (single-element key ⇒ identical behavior).
|
|
1122
|
+
*
|
|
1123
|
+
* A binary-with-replicates consumer (e.g. swe-arena, whose ship-gate counts an
|
|
1124
|
+
* instance resolved only when EVERY replicate resolved) passes a fail-closed
|
|
1125
|
+
* key built from the SAME reduction its gate uses, so winner-selection and the
|
|
1126
|
+
* ship-gate rank on the identical metric and can never invert — the selector
|
|
1127
|
+
* cannot promote a flaky per-cell-mean candidate the gate would reject over a
|
|
1128
|
+
* fail-closed candidate the gate would accept. Only the winner CHOICE changes;
|
|
1129
|
+
* the descriptive `composite` (mean) on every record and the Pareto objective
|
|
1130
|
+
* vectors are untouched, so proposer diversity and reporting are unaffected.
|
|
1131
|
+
*/
|
|
1132
|
+
selectionRankKey?: (campaign: CampaignResult<TArtifact, TScenario>) => number[];
|
|
1133
|
+
/**
|
|
1134
|
+
* Optional policy for which scored surface the next generation MUTATES.
|
|
1135
|
+
* Absent, every generation mutates the global incumbent, so the recorded
|
|
1136
|
+
* `parentSurfaceHash` lineage is a chain. Present, the selector receives the
|
|
1137
|
+
* Pareto frontier so far, the measured incumbent, the generation history,
|
|
1138
|
+
* and the generation index, and returns one frontier parent; the loop hands
|
|
1139
|
+
* that parent to `propose()` as `currentSurface` + `parentOutcome` and
|
|
1140
|
+
* records it as every candidate's `parentSurfaceHash`. Promotion is
|
|
1141
|
+
* unchanged: a candidate still has to beat the incumbent. The loop refuses
|
|
1142
|
+
* a parent it has not measured to completion. `crowdedFrontierParent` is
|
|
1143
|
+
* the provided seeded policy.
|
|
1144
|
+
*/
|
|
1145
|
+
selectParent?: ParentSelector;
|
|
1146
|
+
/**
|
|
1147
|
+
* Record this search into a durable `SearchLedger`. The loop emits the plan,
|
|
1148
|
+
* each candidate-generation operation, each candidate registration with its
|
|
1149
|
+
* measured parent, one task attempt per designed cell, one decision per
|
|
1150
|
+
* candidate, and the terminal event, then returns a bounded
|
|
1151
|
+
* `searchHistory` receipt over the exact ledger bytes.
|
|
1152
|
+
*
|
|
1153
|
+
* `identity` declares what the ledger requires and a campaign cannot infer:
|
|
1154
|
+
* immutable revisions for the agent, proposer, and search implementations,
|
|
1155
|
+
* and the model the agent runs when a cell reports none.
|
|
1156
|
+
*/
|
|
1157
|
+
searchLedger?: SearchLedgerBinding;
|
|
1158
|
+
}
|
|
1159
|
+
type RunOptimizationOptions<TScenario extends Scenario, TArtifact> = RunOptimizationBaseOptions<TScenario, TArtifact>;
|
|
1160
|
+
interface RunOptimizationResult<TArtifact, TScenario extends Scenario> {
|
|
1161
|
+
generations: Array<{
|
|
1162
|
+
record: GenerationRecord;
|
|
1163
|
+
surfaces: Array<{
|
|
1164
|
+
surfaceHash: string;
|
|
1165
|
+
surface: MutableSurface;
|
|
1166
|
+
campaign: CampaignResult<TArtifact, TScenario>;
|
|
1167
|
+
}>;
|
|
1168
|
+
}>;
|
|
1169
|
+
/** Frozen snapshot of the exact starting surface measured by `baselineCampaign`. */
|
|
1170
|
+
baselineSurface: MutableSurface;
|
|
1171
|
+
winnerSurface: MutableSurface;
|
|
1172
|
+
winnerSurfaceHash: string;
|
|
1173
|
+
/** Proposer label for the promoted surface. Present when the winning
|
|
1174
|
+
* candidate came from a `ProposedCandidate` (a reflective proposer);
|
|
1175
|
+
* absent when the winner is the baseline or a bare-surface mutator. */
|
|
1176
|
+
winnerLabel?: string;
|
|
1177
|
+
/** Proposer rationale for the promoted surface — the "because Z" that
|
|
1178
|
+
* motivated the winning change. Survives to `SelfImproveResult` and the
|
|
1179
|
+
* emitted provenance record. Absent when the winner is the baseline. */
|
|
1180
|
+
winnerRationale?: string;
|
|
1181
|
+
baselineCampaign: CampaignResult<TArtifact, TScenario>;
|
|
1182
|
+
/** Run-wide spend, including agents, proposers, analysts, and judges. */
|
|
1183
|
+
cost: CostLedgerSummary;
|
|
1184
|
+
/** Bounded proof envelope over the canonical search ledger. Present only
|
|
1185
|
+
* when `searchLedger` was supplied. `complete` is false when the search was
|
|
1186
|
+
* interrupted or a candidate left a designed cell unscored. */
|
|
1187
|
+
searchHistory?: SearchHistoryReceipt;
|
|
1188
|
+
/** The GEPA Pareto frontier across every scored surface (baseline + all
|
|
1189
|
+
* generations) by per-scenario objective vector — the non-dominated set.
|
|
1190
|
+
* Each generation's `propose()` received the frontier-so-far as
|
|
1191
|
+
* `ctx.paretoParents`; this is the final frontier. A surface here that is
|
|
1192
|
+
* NOT the winner is uniquely best on some scenario the winner loses on. */
|
|
1193
|
+
paretoFrontier: ParetoParent[];
|
|
1194
|
+
}
|
|
1195
|
+
/**
|
|
1196
|
+
* Improvement loop body: N generations of propose → campaign → rank, maintaining a Pareto frontier and one global incumbent across generations. The parent each generation mutates is the incumbent unless `selectParent` draws it from the frontier.
|
|
1197
|
+
*/
|
|
1198
|
+
declare function runOptimization<TScenario extends Scenario, TArtifact>(opts: RunOptimizationOptions<TScenario, TArtifact>): Promise<RunOptimizationResult<TArtifact, TScenario>>;
|
|
1199
|
+
//#endregion
|
|
1200
|
+
//#region src/campaign/presets/run-improvement-loop.d.ts
|
|
1201
|
+
type RunImprovementLoopOptions<TScenario extends Scenario, TArtifact> = RunOptimizationOptions<TScenario, TArtifact> & {
|
|
1202
|
+
/** Holdout scenarios kept OUT of the training optimization pool — used
|
|
1203
|
+
* ONLY to score baseline vs winner for the gate. */
|
|
1204
|
+
holdoutScenarios: TScenario[];
|
|
1205
|
+
/** Holdout policy. Default `'measured'`: baseline + winner are re-scored on
|
|
1206
|
+
* `holdoutScenarios` and the gate decides on that held-out comparison.
|
|
1207
|
+
* `'deferred'`: the improvement-set (search) campaigns run exactly as usual,
|
|
1208
|
+
* but ZERO holdout cells are dispatched, the gate is forced to `'hold'`, and
|
|
1209
|
+
* the result + provenance record carry `holdout: 'deferred'` with NO
|
|
1210
|
+
* held-out lift — for callers that measure the held-out comparison in a
|
|
1211
|
+
* separate later run instead of faking a static holdout scenario and
|
|
1212
|
+
* recording a meaningless lift. */
|
|
1213
|
+
holdout?: 'measured' | 'deferred';
|
|
1214
|
+
/** Promotion gate. Substrate strongly recommends `defaultProductionGate`
|
|
1215
|
+
* for production wiring (composes red-team / reward-hacking / canary /
|
|
1216
|
+
* heldout). */
|
|
1217
|
+
gate: Gate<TArtifact, TScenario>;
|
|
1218
|
+
/** What to do when the gate ships:
|
|
1219
|
+
* - `'pr'`: open a PR via `openAutoPr`
|
|
1220
|
+
* - `'none'`: just report — caller decides what to do with the winner
|
|
1221
|
+
* Live-runtime self-mutation is intentionally unsupported. */
|
|
1222
|
+
autoOnPromote: 'pr' | 'none';
|
|
1223
|
+
/** GH owner / repo for the auto-PR. Required when autoOnPromote === 'pr'. */
|
|
1224
|
+
ghOwner?: string;
|
|
1225
|
+
ghRepo?: string;
|
|
1226
|
+
/** Placebo control. When supplied AND the winner differs from baseline, the
|
|
1227
|
+
* loop scores a THIRD holdout arm: the winner surface with its content
|
|
1228
|
+
* footprint-matched-blanked by this function (typically via `neutralizeText`).
|
|
1229
|
+
* Its scores are exposed to the gate as `ctx.neutralizedJudgeScores`, letting
|
|
1230
|
+
* a `neutralizationGate` reject a win whose lift survives blanking the content
|
|
1231
|
+
* (decorative — driven by footprint, not content). Costs one extra holdout
|
|
1232
|
+
* campaign; omit to skip. Return a byte/layout-matched blank of the winner. */
|
|
1233
|
+
neutralize?: (winnerSurface: MutableSurface, baselineSurface: MutableSurface) => MutableSurface;
|
|
1234
|
+
};
|
|
1235
|
+
interface RunImprovementLoopResult<TArtifact, TScenario extends Scenario> extends RunOptimizationResult<TArtifact, TScenario> {
|
|
1236
|
+
baselineOnHoldout: CampaignResult<TArtifact, TScenario>;
|
|
1237
|
+
winnerOnHoldout: CampaignResult<TArtifact, TScenario>;
|
|
1238
|
+
neutralizedOnHoldout?: CampaignResult<TArtifact, TScenario>;
|
|
1239
|
+
neutralizedSurface?: MutableSurface;
|
|
1240
|
+
gateResult: Awaited<ReturnType<Gate<TArtifact, TScenario>['decide']>>;
|
|
1241
|
+
/** Present iff the loop ran with `holdout: 'deferred'`. When set,
|
|
1242
|
+
* `baselineOnHoldout`/`winnerOnHoldout` are the shared EMPTY campaign (zero
|
|
1243
|
+
* cells dispatched) and the gate verdict is the forced `'hold'`. */
|
|
1244
|
+
holdout?: 'deferred';
|
|
1245
|
+
/** Unified baseline→winner surface diff. Computed UNCONDITIONALLY (not only
|
|
1246
|
+
* when `autoOnPromote === 'pr'`) so the diff that the gate decided on is
|
|
1247
|
+
* always present on the result + in the emitted provenance record. Empty
|
|
1248
|
+
* string when winner == baseline (no change to diff). */
|
|
1249
|
+
promotedDiff: string;
|
|
1250
|
+
prResult?: ReturnType<typeof openAutoPr>;
|
|
1251
|
+
}
|
|
1252
|
+
/**
|
|
1253
|
+
* Gated-promotion shell over `runOptimization`: scores the winner against the baseline on a holdout set, runs the release gate, and optionally opens a PR.
|
|
1254
|
+
*/
|
|
1255
|
+
declare function runImprovementLoop<TScenario extends Scenario, TArtifact>(opts: RunImprovementLoopOptions<TScenario, TArtifact>): Promise<RunImprovementLoopResult<TArtifact, TScenario>>;
|
|
1256
|
+
//#endregion
|
|
1257
|
+
//#region src/campaign/transient-failure.d.ts
|
|
1258
|
+
interface TransientFailureOptions {
|
|
1259
|
+
/**
|
|
1260
|
+
* Treat full-duration timeouts ("timeout after 180000ms") as transient.
|
|
1261
|
+
* Enable on saturated shared infrastructure where queue starvation eats
|
|
1262
|
+
* the clock; leave off when the agent had the resources and simply failed.
|
|
1263
|
+
* Default false.
|
|
1264
|
+
*/
|
|
1265
|
+
readonly retryFullDurationTimeouts?: boolean;
|
|
1266
|
+
/** Additional caller-specific transient patterns. */
|
|
1267
|
+
readonly extraPatterns?: readonly RegExp[];
|
|
1268
|
+
/**
|
|
1269
|
+
* The instant a dated quota refusal is measured against. Defaults to `Date.now()`; inject it
|
|
1270
|
+
* to replay a past classification, which is what a retry audit needs.
|
|
1271
|
+
*/
|
|
1272
|
+
readonly now?: number;
|
|
1273
|
+
}
|
|
1274
|
+
/**
|
|
1275
|
+
* The instant a provider says a spent quota works again, or null when the text states none.
|
|
1276
|
+
*
|
|
1277
|
+
* MEASURED (2026-09-01, discovery lab). The codex/ChatGPT backend answered
|
|
1278
|
+
* `You've hit your usage limit. Visit https://chatgpt.com/codex/settings/usage to purchase more
|
|
1279
|
+
* credits or try again at Sep 6th, 2026 8:29 PM.` — a refusal SIX DAYS out. 25 supervised runs met
|
|
1280
|
+
* it. 21 of them retried it 12 times over about 31 minutes and settled with zero children, zero
|
|
1281
|
+
* tokens and zero claims: about 11 hours of one subscription's capacity spent on a wall that had
|
|
1282
|
+
* already told the caller when it would come down.
|
|
1283
|
+
*
|
|
1284
|
+
* The distinction this draws is not "quota" versus "not quota". It is "the provider named a
|
|
1285
|
+
* release time" versus "it did not". A bare 429, or z.ai's `您的账户已达到速率限制`, recovers on
|
|
1286
|
+
* its own in seconds and SHOULD be retried; those return null here and keep their existing
|
|
1287
|
+
* treatment. Only a stated release is terminal, and only until that instant.
|
|
1288
|
+
*
|
|
1289
|
+
* A date with no zone is read in the host's zone, because a CLI renders it in the host's zone.
|
|
1290
|
+
* An unparseable date returns null rather than a guess: a caller that stops dispatching must
|
|
1291
|
+
* never do so on a misread string.
|
|
1292
|
+
*
|
|
1293
|
+
* @param message the provider's error text
|
|
1294
|
+
* @returns the stated release time, or null when the text names none
|
|
1295
|
+
*/
|
|
1296
|
+
declare function quotaExhaustedUntil(message: string | null | undefined): Date | null;
|
|
1297
|
+
/**
|
|
1298
|
+
* True when the error text describes an infrastructure hiccup that should be
|
|
1299
|
+
* retried rather than scored. Empty/undefined input is not transient.
|
|
1300
|
+
*/
|
|
1301
|
+
declare function isTransientTransportFailure(message: string | null | undefined, opts?: TransientFailureOptions): boolean;
|
|
1302
|
+
/**
|
|
1303
|
+
* Ready-made `cellRetry.retryable` predicate: true for a dispatch-stage
|
|
1304
|
+
* failure whose error message `isTransientTransportFailure` classifies as an
|
|
1305
|
+
* infrastructure hiccup. A judge-stage failure is never retried here — the
|
|
1306
|
+
* dispatch already produced an artifact, so re-dispatching would score a
|
|
1307
|
+
* different sample. A per-cell dispatch deadline ("dispatch exceeded <N>ms")
|
|
1308
|
+
* is not transient by default; opt in via `extraPatterns` or
|
|
1309
|
+
* `retryFullDurationTimeouts` when queue starvation eats the clock. A provider refusal that
|
|
1310
|
+
* states its own release time is never retried while that time is in the future
|
|
1311
|
+
* (`quotaExhaustedUntil`).
|
|
1312
|
+
*/
|
|
1313
|
+
declare function transientDispatchFailure(opts?: TransientFailureOptions): (failure: CampaignCellFailureReceipt['failure']) => boolean;
|
|
1314
|
+
//#endregion
|
|
1315
|
+
//#region src/llm-judge.d.ts
|
|
1316
|
+
/** A rubric dimension as a bare key or the full `{ key, description }` shape. A
|
|
1317
|
+
* bare string uses the key as its own description. */
|
|
1318
|
+
type LlmJudgeDimension = string | JudgeDimension;
|
|
1319
|
+
interface LlmJudgeOptions<TArtifact, TScenario extends Scenario = Scenario> {
|
|
1320
|
+
/** The injected LLM transport. One `chat()` call per `score()`. Required —
|
|
1321
|
+
* there is no default route, so a misconfigured judge fails at construction,
|
|
1322
|
+
* never silently against the free-tier router. */
|
|
1323
|
+
chat: ChatClient;
|
|
1324
|
+
/** Rubric dimensions the model scores. Each becomes a `[0,1]` field of the
|
|
1325
|
+
* returned `JudgeScore.dimensions`. Defaults to a single `quality` dimension. */
|
|
1326
|
+
dimensions?: LlmJudgeDimension[];
|
|
1327
|
+
/** Model id. Falls back to `chat.defaultModel`; one of the two MUST resolve. */
|
|
1328
|
+
model?: string;
|
|
1329
|
+
/** Explicit scoring revision for opaque transport or renderer changes. */
|
|
1330
|
+
judgeVersion?: string;
|
|
1331
|
+
temperature?: number;
|
|
1332
|
+
maxTokens?: number;
|
|
1333
|
+
/** Composite weights forwarded to `weightedComposite`: a partial map selects
|
|
1334
|
+
* AND weights exactly the named dimensions. Omit for a uniform mean. */
|
|
1335
|
+
weights?: Record<string, number>;
|
|
1336
|
+
/**
|
|
1337
|
+
* How to read a score out of the model's answer.
|
|
1338
|
+
*
|
|
1339
|
+
* `'sampled'` (default) reads the number the model emitted. Discrete grades
|
|
1340
|
+
* tie often, and a tie carries no ranking signal.
|
|
1341
|
+
*
|
|
1342
|
+
* `'expectation'` asks the provider for the log probabilities of the score
|
|
1343
|
+
* token and returns the expected value over the integer grades the model
|
|
1344
|
+
* considered, so two answers that both sample `8` separate by how much mass
|
|
1345
|
+
* sat on `7` and `9`. It requires `scale: 'ten'`: an integer grade is one
|
|
1346
|
+
* token, and a `unit` float is not. `whenUnavailable` decides what happens
|
|
1347
|
+
* when the provider returns no log probabilities, or the grade did not land
|
|
1348
|
+
* in one token: `'fail'` throws, `'sampled'` reads the emitted number and
|
|
1349
|
+
* records `scoringMethod: 'sampled'` on the score.
|
|
1350
|
+
*/
|
|
1351
|
+
scoring?: {
|
|
1352
|
+
method: 'sampled';
|
|
1353
|
+
} | {
|
|
1354
|
+
method: 'expectation';
|
|
1355
|
+
whenUnavailable: 'fail' | 'sampled';
|
|
1356
|
+
};
|
|
1357
|
+
/** Scale the model is prompted to score on, normalized into `[0,1]`:
|
|
1358
|
+
* - `'unit'` (default): the model returns `[0,1]` directly.
|
|
1359
|
+
* - `'ten'`: the model returns `[0,10]`; divided by 10 here.
|
|
1360
|
+
* The prompt is annotated with the expected range either way. */
|
|
1361
|
+
scale?: 'unit' | 'ten';
|
|
1362
|
+
/** Run this judge only on matching scenarios (mirrors `JudgeConfig.appliesTo`). */
|
|
1363
|
+
appliesTo?: (scenario: TScenario) => boolean;
|
|
1364
|
+
/** Render the artifact + scenario into the user message. Default:
|
|
1365
|
+
* pretty-printed JSON of `{ scenario, artifact }`. */
|
|
1366
|
+
renderUser?: (input: {
|
|
1367
|
+
artifact: TArtifact;
|
|
1368
|
+
scenario: TScenario;
|
|
1369
|
+
}) => string;
|
|
1370
|
+
/** Strict runtime contract; its JSON Schema is sent to the provider. */
|
|
1371
|
+
costLedger?: CostLedgerHandle;
|
|
1372
|
+
responseSchema?: {
|
|
1373
|
+
name: string;
|
|
1374
|
+
schema: z.ZodObject;
|
|
1375
|
+
};
|
|
1376
|
+
}
|
|
1377
|
+
/**
|
|
1378
|
+
* Build a campaign-shaped `JudgeConfig` whose `score()` makes ONE LLM call
|
|
1379
|
+
* against `prompt` and reduces the model's per-dimension scores to a canonical
|
|
1380
|
+
* `JudgeScore` in `[0,1]`.
|
|
1381
|
+
*
|
|
1382
|
+
* The model is instructed to return JSON `{ "dimensions": { <key>: <number>, … },
|
|
1383
|
+
* "notes": "…" }`; the helper strips fenced JSON, validates every declared
|
|
1384
|
+
* dimension is present and in range, normalizes by `scale`, and composites via
|
|
1385
|
+
* `weightedComposite`.
|
|
1386
|
+
*/
|
|
1387
|
+
declare function llmJudge<TArtifact = unknown, TScenario extends Scenario = Scenario>(name: string, prompt: string, opts: LlmJudgeOptions<TArtifact, TScenario>): JudgeConfig<TArtifact, TScenario>;
|
|
1388
|
+
//#endregion
|
|
1389
|
+
//#region src/reference-equivalence-judge.d.ts
|
|
1390
|
+
declare const REFERENCE_EQUIVALENCE_JUDGE_VERSION = "reference-equivalence-judge-v1-2026-07-13";
|
|
1391
|
+
declare const REFERENCE_EQUIVALENCE_INPUT_LIMITS: {
|
|
1392
|
+
readonly userRequest: 8000;
|
|
1393
|
+
readonly expectedAnswer: 32000;
|
|
1394
|
+
readonly candidateOutput: 32000;
|
|
1395
|
+
};
|
|
1396
|
+
interface ReferenceEquivalenceScenario extends Scenario {
|
|
1397
|
+
userRequest: string;
|
|
1398
|
+
expectedAnswer: string;
|
|
1399
|
+
}
|
|
1400
|
+
interface ReferenceEquivalenceJudgeInput {
|
|
1401
|
+
userRequest: string;
|
|
1402
|
+
expectedAnswer: string;
|
|
1403
|
+
candidateOutput: string;
|
|
1404
|
+
}
|
|
1405
|
+
interface ReferenceEquivalenceJudgeOptions {
|
|
1406
|
+
/** Injected transport. No implicit provider or credentials are selected. */
|
|
1407
|
+
chat: ChatClient;
|
|
1408
|
+
/** Falls back to the ChatClient's default model. */
|
|
1409
|
+
model?: string;
|
|
1410
|
+
/** Used only by the direct-call adapter. */
|
|
1411
|
+
signal?: AbortSignal;
|
|
1412
|
+
/** Optional receipt destination for direct calls; campaigns supply their own. */
|
|
1413
|
+
costLedger?: CostLedgerHandle;
|
|
1414
|
+
}
|
|
1415
|
+
interface ReferenceEquivalenceJudgeResult extends LlmCallMetadata {
|
|
1416
|
+
kind: 'reference-equivalence';
|
|
1417
|
+
version: string;
|
|
1418
|
+
score: number;
|
|
1419
|
+
rationale: string;
|
|
1420
|
+
}
|
|
1421
|
+
/** Build the campaign-native expected-answer judge. */
|
|
1422
|
+
declare function createReferenceEquivalenceJudge(options: ReferenceEquivalenceJudgeOptions): JudgeConfig<string, ReferenceEquivalenceScenario>;
|
|
1423
|
+
/** Direct-call adapter over the campaign judge for product callers. */
|
|
1424
|
+
declare function runReferenceEquivalenceJudge(input: ReferenceEquivalenceJudgeInput, options: ReferenceEquivalenceJudgeOptions): Promise<ReferenceEquivalenceJudgeResult>;
|
|
1425
|
+
//#endregion
|
|
1426
|
+
//#region src/campaign/external-optimizer-observations.d.ts
|
|
1427
|
+
interface ExternalOptimizerObservationSummary {
|
|
1428
|
+
scope: 'callback-submitted-candidates';
|
|
1429
|
+
path: string;
|
|
1430
|
+
sha256: `sha256:${string}`;
|
|
1431
|
+
submittedCandidates: number;
|
|
1432
|
+
evaluations: number;
|
|
1433
|
+
refusals: number;
|
|
1434
|
+
}
|
|
1435
|
+
interface ExternalOptimizerExecutionSummary {
|
|
1436
|
+
scope: 'runtime-model-calls';
|
|
1437
|
+
path: string;
|
|
1438
|
+
sha256: `sha256:${string}`;
|
|
1439
|
+
calls: number;
|
|
1440
|
+
succeeded: number;
|
|
1441
|
+
failed: number;
|
|
1442
|
+
}
|
|
1443
|
+
interface ExternalOptimizerSubmittedCandidate {
|
|
1444
|
+
/** Exact text or named-component surface submitted to the evaluation callback. */
|
|
1445
|
+
readonly candidate: ExternalTextCandidate;
|
|
1446
|
+
/** Eval's canonical content identity for `candidate`. */
|
|
1447
|
+
readonly candidateHash: string;
|
|
1448
|
+
readonly candidateDigest: `sha256:${string}`;
|
|
1449
|
+
readonly proposalSequence: number;
|
|
1450
|
+
/** Exact observation artifact that proves this candidate was submitted. */
|
|
1451
|
+
readonly provenance: {
|
|
1452
|
+
readonly path: string;
|
|
1453
|
+
readonly sha256: `sha256:${string}`;
|
|
1454
|
+
};
|
|
1455
|
+
}
|
|
1456
|
+
interface ExternalOptimizerObservationArtifact {
|
|
1457
|
+
readonly summary: ExternalOptimizerObservationSummary;
|
|
1458
|
+
readonly observations: readonly ExternalOptimizerEvaluationObservation[];
|
|
1459
|
+
/** Every distinct callback-submitted candidate in proposal order. */
|
|
1460
|
+
readonly candidates: readonly ExternalOptimizerSubmittedCandidate[];
|
|
1461
|
+
}
|
|
1462
|
+
/**
|
|
1463
|
+
* Read and verify the exact callback observation artifact addressed by method provenance.
|
|
1464
|
+
*
|
|
1465
|
+
* The reader checks the raw SHA-256, canonical JSONL bytes, sequence, candidate
|
|
1466
|
+
* identities, and summary counts before it returns any candidate.
|
|
1467
|
+
* This proves that the bytes match the supplied summary. The caller remains
|
|
1468
|
+
* responsible for obtaining that summary from trusted provenance.
|
|
1469
|
+
*/
|
|
1470
|
+
declare function readExternalOptimizerObservationArtifact(input: {
|
|
1471
|
+
summary: ExternalOptimizerObservationSummary;
|
|
1472
|
+
storage?: CampaignStorage;
|
|
1473
|
+
}): ExternalOptimizerObservationArtifact;
|
|
1474
|
+
//#endregion
|
|
1475
|
+
//#region src/campaign/optimization-cost.d.ts
|
|
1476
|
+
/** Cost reported by a method or by final test scoring. */
|
|
1477
|
+
interface ComparisonCost {
|
|
1478
|
+
/** Known subtotal. Consult `costProvenance` before treating this as total spend. */
|
|
1479
|
+
totalCostUsd: number;
|
|
1480
|
+
/** Exact origin of the total; uncaptured means `totalCostUsd` is only a known subtotal. */
|
|
1481
|
+
costProvenance: CostProvenance;
|
|
1482
|
+
accountingComplete: boolean;
|
|
1483
|
+
incompleteReasons: string[];
|
|
1484
|
+
}
|
|
1485
|
+
/** Keep the cost fields a custom optimization method must report. */
|
|
1486
|
+
declare function costFromLedgerSummary(summary: CostLedgerSummary): ComparisonCost;
|
|
1487
|
+
/** Combine method costs without turning one unknown bill into a known total. */
|
|
1488
|
+
declare function combineComparisonCosts(entries: ReadonlyArray<{
|
|
1489
|
+
label: string;
|
|
1490
|
+
cost: ComparisonCost;
|
|
1491
|
+
}>): ComparisonCost;
|
|
1492
|
+
//#endregion
|
|
1493
|
+
//#region src/campaign/presets/compare-optimization-methods.d.ts
|
|
1494
|
+
/** Shared campaign settings applied to every optimization method. */
|
|
1495
|
+
type OptimizationMethodRunOptions<TScenario extends Scenario, TArtifact> = Omit<RunCampaignOptions<TScenario, TArtifact>, 'costCeiling' | 'costLedger' | 'dispatch' | 'judges' | 'runDir' | 'scenarios' | 'seed'>;
|
|
1496
|
+
interface OptimizationPackageSource {
|
|
1497
|
+
kind: 'package';
|
|
1498
|
+
/** Whether package identity was inspected or supplied by caller code. */
|
|
1499
|
+
evidence: 'observed' | 'declared';
|
|
1500
|
+
package: string;
|
|
1501
|
+
version: string;
|
|
1502
|
+
sourceUrl?: string;
|
|
1503
|
+
revision?: string;
|
|
1504
|
+
/** SHA-256 of all installed module files observed before the run. */
|
|
1505
|
+
sourceSha256?: string;
|
|
1506
|
+
}
|
|
1507
|
+
interface OptimizationModuleSource {
|
|
1508
|
+
module: string;
|
|
1509
|
+
sourceSha256: string;
|
|
1510
|
+
}
|
|
1511
|
+
interface OptimizationPythonRuntime {
|
|
1512
|
+
implementation: string;
|
|
1513
|
+
version: string;
|
|
1514
|
+
}
|
|
1515
|
+
interface OptimizationTokenUsage {
|
|
1516
|
+
/** All input tokens, including cache reads and cache creation. */
|
|
1517
|
+
inputTokens: number;
|
|
1518
|
+
/** Input tokens served from a provider cache. */
|
|
1519
|
+
cachedInputTokens?: number;
|
|
1520
|
+
/** Input tokens used to create or write a provider cache entry. */
|
|
1521
|
+
cacheWriteInputTokens?: number;
|
|
1522
|
+
outputTokens: number;
|
|
1523
|
+
/** Reasoning tokens included in `outputTokens`. */
|
|
1524
|
+
reasoningTokens?: number;
|
|
1525
|
+
totalTokens: number;
|
|
1526
|
+
calls: number;
|
|
1527
|
+
}
|
|
1528
|
+
interface OptimizationMethodProvenance {
|
|
1529
|
+
/** External optimizer package. */
|
|
1530
|
+
source: OptimizationPackageSource;
|
|
1531
|
+
/** Python bridge package that invoked the optimizer. */
|
|
1532
|
+
bridge?: OptimizationPackageSource;
|
|
1533
|
+
/** Custom engine modules imported by the optimizer. */
|
|
1534
|
+
modules?: OptimizationModuleSource[];
|
|
1535
|
+
/** Python implementation used by the bridge process. */
|
|
1536
|
+
python?: OptimizationPythonRuntime;
|
|
1537
|
+
/** Exact model identifier configured for optimizer-owned model calls. */
|
|
1538
|
+
optimizerModel?: string;
|
|
1539
|
+
/** Stable public identity of the execution-owner callback. */
|
|
1540
|
+
optimizerCallRef?: string;
|
|
1541
|
+
runId: string;
|
|
1542
|
+
/** Content identity shared by compatible resumptions. */
|
|
1543
|
+
compatibleRunId?: string;
|
|
1544
|
+
resumed: boolean;
|
|
1545
|
+
/** Whether the run seed reached every external engine configuration. */
|
|
1546
|
+
seedApplied?: boolean;
|
|
1547
|
+
/** Evaluations the local callback metered — the trusted count. */
|
|
1548
|
+
evaluationCount: number;
|
|
1549
|
+
/**
|
|
1550
|
+
* Evaluation total the external optimizer reported from its own counters.
|
|
1551
|
+
* A difference from `evaluationCount` means upstream skipped, cached, or
|
|
1552
|
+
* double-counted work; inspect before trusting upstream-derived budgets.
|
|
1553
|
+
*/
|
|
1554
|
+
upstreamReportedEvaluations?: number;
|
|
1555
|
+
artifactDir: string;
|
|
1556
|
+
tokenUsage?: OptimizationTokenUsage;
|
|
1557
|
+
/** Candidates submitted to the callback, per-case scores, and refusals. */
|
|
1558
|
+
observations?: ExternalOptimizerObservationSummary;
|
|
1559
|
+
/** Exact accepted GEPA candidates, parent indices, and selection scores. */
|
|
1560
|
+
gepaCandidatePopulation?: GepaCandidatePopulationSummary;
|
|
1561
|
+
/** Opaque Runtime execution evidence for every invoked optimizer-model call. */
|
|
1562
|
+
modelExecutions?: ExternalOptimizerExecutionSummary;
|
|
1563
|
+
/** Anthropic-endpoint proxy traffic from agent CLI engines, when enabled. */
|
|
1564
|
+
anthropicEndpoint?: ExternalOptimizerWireCounts;
|
|
1565
|
+
}
|
|
1566
|
+
/** Shared inputs for one optimization method. Final test data is absent. */
|
|
1567
|
+
interface OptimizationMethodInput<TScenario extends Scenario, TArtifact> {
|
|
1568
|
+
/** Surface every method starts from. */
|
|
1569
|
+
readonly baselineSurface: MutableSurface;
|
|
1570
|
+
/** Evidence used to author or fit candidates. */
|
|
1571
|
+
readonly trainScenarios: readonly TScenario[];
|
|
1572
|
+
/** Data used for candidate acceptance, early stopping, and model selection. */
|
|
1573
|
+
readonly selectionScenarios: readonly TScenario[];
|
|
1574
|
+
/** Runs one scenario with a candidate surface. */
|
|
1575
|
+
readonly dispatchWithSurface: (surface: MutableSurface, scenario: TScenario, ctx: DispatchContext) => Promise<TArtifact>;
|
|
1576
|
+
/** Scores artifacts produced by `dispatchWithSurface`. */
|
|
1577
|
+
readonly judges: readonly JudgeConfig<TArtifact, TScenario>[];
|
|
1578
|
+
/** Method-specific artifacts are written below this directory. */
|
|
1579
|
+
readonly runDir: string;
|
|
1580
|
+
readonly seed: number;
|
|
1581
|
+
/** Shared defaults for every method. A method may override them explicitly. */
|
|
1582
|
+
readonly runOptions: Readonly<OptimizationMethodRunOptions<TScenario, TArtifact>>;
|
|
1583
|
+
/** Durable spend account shared by every method and final scoring. */
|
|
1584
|
+
readonly costLedger: CostLedgerHandle;
|
|
1585
|
+
}
|
|
1586
|
+
interface OptimizationMethodResult {
|
|
1587
|
+
/** Surface selected without using the final test partition. */
|
|
1588
|
+
winnerSurface: MutableSurface;
|
|
1589
|
+
/** Optimization spend. Excludes final test scoring. */
|
|
1590
|
+
cost: ComparisonCost;
|
|
1591
|
+
/** Optimization duration. Excludes final test scoring. */
|
|
1592
|
+
durationMs?: number;
|
|
1593
|
+
/** Exact external implementation and run identity, when the method uses one. */
|
|
1594
|
+
provenance?: OptimizationMethodProvenance;
|
|
1595
|
+
/** Bounded proof envelope over the canonical SearchLedger for this optimization. */
|
|
1596
|
+
searchHistory?: SearchHistoryReceipt;
|
|
1597
|
+
}
|
|
1598
|
+
/** A complete optimization method, including candidate generation and selection. */
|
|
1599
|
+
interface OptimizationMethod<TScenario extends Scenario = Scenario, TArtifact = unknown> {
|
|
1600
|
+
/** Unique, trimmed display name. Its normalized form must also be unique. */
|
|
1601
|
+
name: string;
|
|
1602
|
+
optimize: (input: OptimizationMethodInput<TScenario, TArtifact>) => Promise<OptimizationMethodResult>;
|
|
1603
|
+
}
|
|
1604
|
+
interface OptimizationMethodScore {
|
|
1605
|
+
name: string;
|
|
1606
|
+
/** Mean final-test composite of the baseline (identical across methods). */
|
|
1607
|
+
baselineComposite: number;
|
|
1608
|
+
/** Mean final-test composite of this method's selected surface. */
|
|
1609
|
+
winnerComposite: number;
|
|
1610
|
+
/** Mean per-scenario final-test lift (winner minus baseline). */
|
|
1611
|
+
lift: number;
|
|
1612
|
+
/** Simultaneous paired-bootstrap interval for per-scenario lift.
|
|
1613
|
+
* `low > 0` excludes zero after adjustment for all reported contrasts. */
|
|
1614
|
+
liftCi: {
|
|
1615
|
+
low: number;
|
|
1616
|
+
high: number;
|
|
1617
|
+
};
|
|
1618
|
+
/** Search spend reconciled with recorded method calls. Excludes final test scoring. */
|
|
1619
|
+
optimizationCost: ComparisonCost;
|
|
1620
|
+
/** Optimization duration reported by the method. Excludes final test scoring. */
|
|
1621
|
+
durationMs?: number;
|
|
1622
|
+
/** Exact external implementation and run identity, when reported by the method. */
|
|
1623
|
+
provenance?: OptimizationMethodProvenance;
|
|
1624
|
+
/** Paired final-test values used to compute lift and its interval. */
|
|
1625
|
+
scenarioScores: Array<{
|
|
1626
|
+
scenarioId: string;
|
|
1627
|
+
baselineComposite: number;
|
|
1628
|
+
winnerComposite: number;
|
|
1629
|
+
lift: number;
|
|
1630
|
+
}>;
|
|
1631
|
+
winnerSurface: MutableSurface;
|
|
1632
|
+
/** 1-based, by descending lift. */
|
|
1633
|
+
rank: number;
|
|
1634
|
+
}
|
|
1635
|
+
interface OptimizationMethodPairwise {
|
|
1636
|
+
/** Higher-ranked method. */
|
|
1637
|
+
a: string;
|
|
1638
|
+
b: string;
|
|
1639
|
+
/** Mean per-scenario untouched-test delta (a − b). */
|
|
1640
|
+
deltaMean: number;
|
|
1641
|
+
low: number;
|
|
1642
|
+
high: number;
|
|
1643
|
+
/** `a` if the CI clears 0, `b` if it is entirely negative, else `'tie'`. */
|
|
1644
|
+
favored: string;
|
|
1645
|
+
}
|
|
1646
|
+
interface OptimizationMethodComparison {
|
|
1647
|
+
/** Sorted by descending lift; `rank` set accordingly. */
|
|
1648
|
+
scores: OptimizationMethodScore[];
|
|
1649
|
+
best: OptimizationMethodScore;
|
|
1650
|
+
/** Best vs each other method, using simultaneous paired-bootstrap intervals. */
|
|
1651
|
+
pairwise: OptimizationMethodPairwise[];
|
|
1652
|
+
testScenarioIds: string[];
|
|
1653
|
+
/** Sum of method reports reconciled against each method's recorded calls. */
|
|
1654
|
+
optimizationCost: ComparisonCost;
|
|
1655
|
+
/** Baseline and distinct winner scoring on the final test partition. */
|
|
1656
|
+
testCost: ComparisonCost;
|
|
1657
|
+
/** Optimization plus final test scoring. */
|
|
1658
|
+
totalCost: ComparisonCost;
|
|
1659
|
+
/** Caller-requested simultaneous coverage across all reported contrasts. */
|
|
1660
|
+
confidence: number;
|
|
1661
|
+
/** Bonferroni-adjusted confidence used for each bootstrap interval. */
|
|
1662
|
+
intervalConfidence: number;
|
|
1663
|
+
/** Method-vs-baseline plus all possible method-vs-method contrasts. */
|
|
1664
|
+
comparisonCount: number;
|
|
1665
|
+
/** Deterministic bootstrap and campaign seed. */
|
|
1666
|
+
seed: number;
|
|
1667
|
+
/** Bootstrap draws used for each interval. */
|
|
1668
|
+
resamples: number;
|
|
1669
|
+
/** Agent runs averaged within each test scenario before resampling scenarios. */
|
|
1670
|
+
reps: number;
|
|
1671
|
+
/** Coverage of every method's canonical search history. */
|
|
1672
|
+
searchHistory: SearchHistoryCoverage;
|
|
1673
|
+
}
|
|
1674
|
+
interface CompareOptimizationMethodsOptions<TScenario extends Scenario, TArtifact> extends Omit<RunCampaignOptions<TScenario, TArtifact>, 'dispatch' | 'judges' | 'scenarios'> {
|
|
1675
|
+
methods: OptimizationMethod<TScenario, TArtifact>[];
|
|
1676
|
+
baselineSurface: MutableSurface;
|
|
1677
|
+
/** Evidence used by every optimizer to author or fit candidates. */
|
|
1678
|
+
trainScenarios: TScenario[];
|
|
1679
|
+
/** Candidate acceptance, early-stopping, and optimizer-selection data. */
|
|
1680
|
+
selectionScenarios: TScenario[];
|
|
1681
|
+
/** Untouched final comparison data. Never passed to an optimization method. */
|
|
1682
|
+
testScenarios: TScenario[];
|
|
1683
|
+
/** Scores a surface on a scenario. The methods and final test share this function. */
|
|
1684
|
+
dispatchWithSurface: (surface: MutableSurface, scenario: TScenario, ctx: DispatchContext) => Promise<TArtifact>;
|
|
1685
|
+
judges: JudgeConfig<TArtifact, TScenario>[];
|
|
1686
|
+
/** Bootstrap resamples for the lift intervals. Default is at least 2000 and
|
|
1687
|
+
* rises when the requested simultaneous confidence needs finer tails. */
|
|
1688
|
+
resamples?: number;
|
|
1689
|
+
/** Shared defaults for each method's train and selection campaigns. */
|
|
1690
|
+
optimizationRunOptions?: OptimizationMethodRunOptions<TScenario, TArtifact>;
|
|
1691
|
+
/** Number of optimization methods to run concurrently. Default 1. */
|
|
1692
|
+
optimizationConcurrency?: number;
|
|
1693
|
+
/** Simultaneous confidence across method-vs-baseline and method-vs-method contrasts.
|
|
1694
|
+
* Each bootstrap interval is Bonferroni-adjusted. Default 0.95. */
|
|
1695
|
+
confidence?: number;
|
|
1696
|
+
/** Shared spend limit across every method's optimizer and evaluation calls plus final scoring. */
|
|
1697
|
+
costCeiling?: number;
|
|
1698
|
+
/**
|
|
1699
|
+
* Missing history is reported by default. Publication-grade or autonomous
|
|
1700
|
+
* callers set `require-complete`, which aborts before the first final-test call.
|
|
1701
|
+
*/
|
|
1702
|
+
searchHistoryPolicy?: SearchHistoryPolicy;
|
|
1703
|
+
}
|
|
1704
|
+
/**
|
|
1705
|
+
* Compare complete optimization methods on disjoint train, selection, and final test data.
|
|
1706
|
+
*/
|
|
1707
|
+
declare function compareOptimizationMethods<TScenario extends Scenario, TArtifact>(opts: CompareOptimizationMethodsOptions<TScenario, TArtifact>): Promise<OptimizationMethodComparison>;
|
|
1708
|
+
/** Preserve every optimizer token class while keeping total input and output explicit. */
|
|
1709
|
+
declare function optimizationTokenUsageFromSummary(summary: CostLedgerSummary, receipts: readonly CostReceipt[]): OptimizationTokenUsage | undefined;
|
|
1710
|
+
//#endregion
|
|
1711
|
+
//#region src/campaign/external-text-evaluation.d.ts
|
|
1712
|
+
interface ExternalOptimizationExample {
|
|
1713
|
+
id: string;
|
|
1714
|
+
data: unknown;
|
|
1715
|
+
}
|
|
1716
|
+
interface ExternalTextEvaluationResponse {
|
|
1717
|
+
score: number;
|
|
1718
|
+
info: {
|
|
1719
|
+
scenarioId: string;
|
|
1720
|
+
dimensions: Record<string, number>;
|
|
1721
|
+
notes?: string;
|
|
1722
|
+
artifact?: unknown;
|
|
1723
|
+
/** The optimizer score is a finite protocol value; Eval remains the
|
|
1724
|
+
* authority for whether the underlying cell actually produced a score. */
|
|
1725
|
+
status?: 'scored' | 'failed';
|
|
1726
|
+
error?: {
|
|
1727
|
+
stage: 'dispatch' | 'judge' | 'unknown';
|
|
1728
|
+
message: string;
|
|
1729
|
+
};
|
|
1730
|
+
};
|
|
1731
|
+
}
|
|
1732
|
+
declare function decodeExternalTextCandidate(candidate: ExternalTextCandidate): MutableSurface;
|
|
1733
|
+
//#endregion
|
|
1734
|
+
//#region src/campaign/external-text-optimization-contract.d.ts
|
|
1735
|
+
interface ExternalTextOptimizerContext {
|
|
1736
|
+
readonly runId: string;
|
|
1737
|
+
readonly name: string;
|
|
1738
|
+
readonly objective: string;
|
|
1739
|
+
readonly evaluationId: string;
|
|
1740
|
+
readonly background?: string;
|
|
1741
|
+
readonly seedCandidate: ExternalTextCandidate;
|
|
1742
|
+
readonly trainSet: readonly ExternalOptimizationExample[];
|
|
1743
|
+
readonly selectionSet: readonly ExternalOptimizationExample[];
|
|
1744
|
+
readonly maxEvaluations: number;
|
|
1745
|
+
readonly seed: number;
|
|
1746
|
+
/** Stable directory for optimizer checkpoints from compatible attempts. */
|
|
1747
|
+
readonly stateDir: string;
|
|
1748
|
+
readonly restoreRequested: boolean;
|
|
1749
|
+
readonly artifactDir: string;
|
|
1750
|
+
readonly signal: AbortSignal;
|
|
1751
|
+
/** Record every optimizer-owned paid call through this attributed account. */
|
|
1752
|
+
readonly cost: CampaignCostMeter;
|
|
1753
|
+
readonly evaluate: (request: ExternalTextEvaluationRequest) => Promise<ExternalTextEvaluationResponse>;
|
|
1754
|
+
}
|
|
1755
|
+
interface ExternalTextOptimizerResult {
|
|
1756
|
+
bestCandidate: ExternalTextCandidate;
|
|
1757
|
+
resumed: boolean;
|
|
1758
|
+
costAccounting: {
|
|
1759
|
+
kind: 'metered';
|
|
1760
|
+
} | {
|
|
1761
|
+
kind: 'no-paid-work';
|
|
1762
|
+
} | {
|
|
1763
|
+
kind: 'external';
|
|
1764
|
+
reason: string;
|
|
1765
|
+
};
|
|
1766
|
+
}
|
|
1767
|
+
/**
|
|
1768
|
+
* Configuration for adapting another text optimizer.
|
|
1769
|
+
*
|
|
1770
|
+
* `run` owns search. Agent Eval owns split isolation, bounded candidate
|
|
1771
|
+
* evaluation, exact cost collection, provenance, and final comparison.
|
|
1772
|
+
*/
|
|
1773
|
+
interface ExternalTextOptimizationMethodConfig<TScenario extends Scenario, TArtifact = unknown> {
|
|
1774
|
+
name: string;
|
|
1775
|
+
source: Omit<OptimizationPackageSource, 'evidence'>;
|
|
1776
|
+
objective: string;
|
|
1777
|
+
evaluationId: string;
|
|
1778
|
+
background?: string;
|
|
1779
|
+
maxEvaluations: number;
|
|
1780
|
+
/** Hard limit for calls made through `context.cost`. Use 0 for no paid work. */
|
|
1781
|
+
maxOptimizerCostUsd: number;
|
|
1782
|
+
/** Abort `context.signal` after this duration. Default: 30 minutes. */
|
|
1783
|
+
timeoutMs?: number;
|
|
1784
|
+
/** Default: `never`. Compatible runs reuse one state directory. */
|
|
1785
|
+
resume?: ExternalOptimizerResumeMode;
|
|
1786
|
+
maxCandidateChars?: number;
|
|
1787
|
+
maxEvidenceChars?: number;
|
|
1788
|
+
describeScenario?: (scenario: TScenario) => unknown;
|
|
1789
|
+
describeArtifact?: (artifact: TArtifact, scenario: TScenario) => unknown;
|
|
1790
|
+
run: (context: ExternalTextOptimizerContext) => Promise<ExternalTextOptimizerResult>;
|
|
1791
|
+
}
|
|
1792
|
+
//#endregion
|
|
1793
|
+
//#region src/campaign/external-text-optimization.d.ts
|
|
1794
|
+
/**
|
|
1795
|
+
* Adapt a third-party text optimizer without reimplementing its search.
|
|
1796
|
+
*
|
|
1797
|
+
* The callback never receives final test cases. Calls to `evaluate` are
|
|
1798
|
+
* counted before execution and stop at `maxEvaluations`.
|
|
1799
|
+
*/
|
|
1800
|
+
declare function externalTextOptimizationMethod<TScenario extends Scenario, TArtifact>(config: ExternalTextOptimizationMethodConfig<TScenario, TArtifact>): OptimizationMethod<TScenario, TArtifact>;
|
|
1801
|
+
//#endregion
|
|
1802
|
+
//#region src/campaign/optimizer-model.d.ts
|
|
1803
|
+
type OptimizerModelBudget = ExternalOptimizerModelBudget;
|
|
1804
|
+
/** One metered model path supplied by the package that owns execution. */
|
|
1805
|
+
interface OpenAICompatibleOptimizerModel {
|
|
1806
|
+
model: string;
|
|
1807
|
+
budget: OptimizerModelBudget;
|
|
1808
|
+
/** Caller-owned execution path, such as Runtime's exact AgentProfile adapter. */
|
|
1809
|
+
call: ExternalOptimizerModelCall;
|
|
1810
|
+
/** Stable public identity included in resumable-run compatibility. */
|
|
1811
|
+
callRef: string;
|
|
1812
|
+
/**
|
|
1813
|
+
* Served-model acceptance for every proxied call. Default `'exact'`, which
|
|
1814
|
+
* rejects any substitution with a 502 to the child.
|
|
1815
|
+
* `'allow-within-family'` accepts a different model of the same provider
|
|
1816
|
+
* family; it keeps family-level claims valid and forfeits per-model claims.
|
|
1817
|
+
*/
|
|
1818
|
+
servedModelPolicy?: ServedModelPolicy;
|
|
1819
|
+
/**
|
|
1820
|
+
* Also expose this metered model over the Anthropic Messages API
|
|
1821
|
+
* (`POST /v1/messages`) on the loopback proxy. Required for GEPA agent
|
|
1822
|
+
* engines that drive a `claude` CLI subprocess. Agent engines are chatty:
|
|
1823
|
+
* size `budget.maxRequests` for tens of calls per engine run. Default: false.
|
|
1824
|
+
*/
|
|
1825
|
+
anthropicEndpoint?: boolean;
|
|
1826
|
+
}
|
|
1827
|
+
//#endregion
|
|
1828
|
+
//#region src/campaign/gepa-optimization-method.d.ts
|
|
1829
|
+
/** Shared settings for one bounded GEPA engine invocation. */
|
|
1830
|
+
interface GepaEngineOptions {
|
|
1831
|
+
/** GEPA engine name. GEPA validates names available in its Python runtime. */
|
|
1832
|
+
engine: string;
|
|
1833
|
+
/** Optional billed-USD stop. Omit when the execution owner reports USD as unknown. */
|
|
1834
|
+
maxProposerCostUsd?: number;
|
|
1835
|
+
/** Maximum concurrent evaluations inside this engine. Default: 1. */
|
|
1836
|
+
maxConcurrency?: number;
|
|
1837
|
+
/** Stop the engine after it reaches this score. */
|
|
1838
|
+
stopAtScore?: number;
|
|
1839
|
+
/** Isolate agent-based engines. Default: true. */
|
|
1840
|
+
sandbox?: boolean;
|
|
1841
|
+
/**
|
|
1842
|
+
* JSON-safe configuration for the registered GEPA engine.
|
|
1843
|
+
* Python callables and class instances cannot cross the process boundary.
|
|
1844
|
+
*/
|
|
1845
|
+
engineConfig?: Record<string, unknown>;
|
|
1846
|
+
}
|
|
1847
|
+
/** One independently budgeted GEPA engine invocation. */
|
|
1848
|
+
interface GepaEngineRun extends GepaEngineOptions {
|
|
1849
|
+
/** Maximum callback evaluations this engine may consume. */
|
|
1850
|
+
maxEvaluations: number;
|
|
1851
|
+
}
|
|
1852
|
+
/** An engine in an adaptive run. All engines share the recipe evaluation limit. */
|
|
1853
|
+
type GepaAdaptiveEngineRun = GepaEngineOptions;
|
|
1854
|
+
/**
|
|
1855
|
+
* A direct mapping to a GEPA optimization recipe.
|
|
1856
|
+
*
|
|
1857
|
+
* GEPA owns every search and composition operation represented here. Tangle
|
|
1858
|
+
* supplies the candidate, data, execution callback, judges, and budgets.
|
|
1859
|
+
*/
|
|
1860
|
+
type GepaOptimizationRecipe = {
|
|
1861
|
+
kind: 'engine';
|
|
1862
|
+
run: GepaEngineRun;
|
|
1863
|
+
} | {
|
|
1864
|
+
kind: 'sequential';
|
|
1865
|
+
runs: readonly GepaEngineRun[];
|
|
1866
|
+
} | {
|
|
1867
|
+
kind: 'adaptive-sequential';
|
|
1868
|
+
runs: readonly GepaAdaptiveEngineRun[];
|
|
1869
|
+
/** One evaluation budget shared by every adaptive stage. */
|
|
1870
|
+
maxEvaluations: number;
|
|
1871
|
+
/** Switch engines after this many evaluations without improvement. */
|
|
1872
|
+
plateauEvaluations: number;
|
|
1873
|
+
patience?: number;
|
|
1874
|
+
minEvaluationsPerStage?: number;
|
|
1875
|
+
improvementEpsilon?: number;
|
|
1876
|
+
cycle?: boolean;
|
|
1877
|
+
maxSwitches?: number;
|
|
1878
|
+
maxConcurrency?: number;
|
|
1879
|
+
} | {
|
|
1880
|
+
kind: 'best-of';
|
|
1881
|
+
runs: readonly GepaEngineRun[];
|
|
1882
|
+
maxWorkers?: number;
|
|
1883
|
+
} | {
|
|
1884
|
+
kind: 'vote';
|
|
1885
|
+
runs: readonly GepaEngineRun[];
|
|
1886
|
+
maxWorkers?: number;
|
|
1887
|
+
} | {
|
|
1888
|
+
kind: 'omni';
|
|
1889
|
+
explore: readonly GepaEngineRun[];
|
|
1890
|
+
continueWith: GepaEngineRun;
|
|
1891
|
+
maxWorkers?: number;
|
|
1892
|
+
};
|
|
1893
|
+
/** The command that runs the Python GEPA bridge. */
|
|
1894
|
+
type GepaRunnerCommand = ExternalOptimizerRunnerCommand;
|
|
1895
|
+
interface GepaOptimizationMethodConfig<TScenario extends Scenario, TArtifact = unknown> {
|
|
1896
|
+
/** Unique comparison-method name. Default identifies the GEPA recipe. */
|
|
1897
|
+
name?: string;
|
|
1898
|
+
/** A direct GEPA recipe. */
|
|
1899
|
+
recipe: GepaOptimizationRecipe;
|
|
1900
|
+
/** Plain-language goal shown to the external optimizer. */
|
|
1901
|
+
objective: string;
|
|
1902
|
+
/** Stable identity for the dispatch, judges, model settings, and scoring logic. */
|
|
1903
|
+
evaluationId: string;
|
|
1904
|
+
/** Optional bounded context about the surface and task. */
|
|
1905
|
+
background?: string;
|
|
1906
|
+
/**
|
|
1907
|
+
* Public dotted Python modules imported before GEPA resolves engine names.
|
|
1908
|
+
* Each module should call GEPA's official `register_engine()` API at import.
|
|
1909
|
+
*/
|
|
1910
|
+
engineModules?: readonly string[];
|
|
1911
|
+
/** Reject external candidates longer than this. Default: 200,000 characters. */
|
|
1912
|
+
maxCandidateChars?: number;
|
|
1913
|
+
/** Reject serialized score evidence longer than this. Default: 100,000 characters. */
|
|
1914
|
+
maxEvidenceChars?: number;
|
|
1915
|
+
/** Candidate-evaluation callback byte limits. Omitted fields use finite defaults. */
|
|
1916
|
+
evaluationCallbackLimits?: Partial<ExternalOptimizerCallbackLimits>;
|
|
1917
|
+
/** End the bridge process after this many milliseconds. Default: 30 minutes. */
|
|
1918
|
+
timeoutMs?: number;
|
|
1919
|
+
/**
|
|
1920
|
+
* OpenAI-compatible model used by standard GEPA reflection.
|
|
1921
|
+
* Calls pass through Agent Eval's local model proxy. Every recipe engine must
|
|
1922
|
+
* be `gepa` when this is set.
|
|
1923
|
+
*/
|
|
1924
|
+
optimizer?: OpenAICompatibleOptimizerModel;
|
|
1925
|
+
/**
|
|
1926
|
+
* Decide what the external optimizer may read for a train or selection case.
|
|
1927
|
+
* The returned value must be JSON-serializable. The final comparison cases
|
|
1928
|
+
* are not accepted by this API and cannot be serialized here.
|
|
1929
|
+
*/
|
|
1930
|
+
describeScenario?: (scenario: TScenario) => unknown;
|
|
1931
|
+
/** Optional bounded artifact evidence returned to GEPA after each evaluation. */
|
|
1932
|
+
describeArtifact?: (artifact: TArtifact, scenario: TScenario) => unknown;
|
|
1933
|
+
/** Default: `never`. Compatible runs resume only when explicitly enabled. */
|
|
1934
|
+
resume?: ExternalOptimizerResumeMode;
|
|
1935
|
+
/**
|
|
1936
|
+
* Required for resumable direct GEPA runs because upstream state uses Python
|
|
1937
|
+
* pickle. Enable only for state created locally in a directory you control.
|
|
1938
|
+
*/
|
|
1939
|
+
trustResumeState?: boolean;
|
|
1940
|
+
runner?: GepaRunnerCommand;
|
|
1941
|
+
/**
|
|
1942
|
+
* Record GEPA's own candidate population into the canonical `SearchLedger`
|
|
1943
|
+
* and return the bounded receipt on the method result, so a comparison run
|
|
1944
|
+
* under `searchHistoryPolicy: 'require-complete'` accepts this method.
|
|
1945
|
+
*
|
|
1946
|
+
* `identity` declares the immutable revisions and the model snapshot the
|
|
1947
|
+
* ledger requires and the bridge does not report. `path` defaults to
|
|
1948
|
+
* `<runDir>/search-ledger.jsonl`.
|
|
1949
|
+
*/
|
|
1950
|
+
searchLedger?: {
|
|
1951
|
+
identity: SearchRunIdentity;
|
|
1952
|
+
path?: string;
|
|
1953
|
+
};
|
|
1954
|
+
}
|
|
1955
|
+
declare function gepaOptimizationMethod<TScenario extends Scenario, TArtifact>(config: GepaOptimizationMethodConfig<TScenario, TArtifact>): OptimizationMethod<TScenario, TArtifact>;
|
|
1956
|
+
//#endregion
|
|
1957
|
+
//#region src/campaign/skillopt-optimization-method.d.ts
|
|
1958
|
+
interface SkillOptTrainerConfig {
|
|
1959
|
+
epochs: number;
|
|
1960
|
+
batchSize: number;
|
|
1961
|
+
accumulation?: number;
|
|
1962
|
+
editBudget?: number;
|
|
1963
|
+
minEditBudget?: number;
|
|
1964
|
+
analystWorkers?: number;
|
|
1965
|
+
minibatchSize?: number;
|
|
1966
|
+
mergeBatchSize?: number;
|
|
1967
|
+
maxAnalystRounds?: number;
|
|
1968
|
+
evaluationWorkers?: number;
|
|
1969
|
+
learningRateSchedule?: 'constant' | 'linear' | 'cosine' | 'autonomous';
|
|
1970
|
+
learningRateControl?: 'fixed' | 'autonomous' | 'none';
|
|
1971
|
+
updateMode?: 'patch' | 'rewrite_from_suggestions' | 'full_rewrite_minibatch';
|
|
1972
|
+
failureOnly?: boolean;
|
|
1973
|
+
useSlowUpdate?: boolean;
|
|
1974
|
+
useMetaSkill?: boolean;
|
|
1975
|
+
/**
|
|
1976
|
+
* Additional flat SkillOpt trainer settings. Tangle overwrites data,
|
|
1977
|
+
* output, split, seed, validation, and activation settings.
|
|
1978
|
+
*/
|
|
1979
|
+
overrides?: Record<string, unknown>;
|
|
1980
|
+
}
|
|
1981
|
+
type SkillOptRunnerCommand = ExternalOptimizerRunnerCommand;
|
|
1982
|
+
interface SkillOptOptimizationMethodConfig<TScenario extends Scenario, TArtifact = unknown> {
|
|
1983
|
+
name?: string;
|
|
1984
|
+
/** Goal included with every described train and selection case. */
|
|
1985
|
+
objective: string;
|
|
1986
|
+
background?: string;
|
|
1987
|
+
/** Stable identity for the dispatch, judges, model settings, and scoring logic. */
|
|
1988
|
+
evaluationId: string;
|
|
1989
|
+
trainer: SkillOptTrainerConfig;
|
|
1990
|
+
/**
|
|
1991
|
+
* OpenAI-compatible model connection and hard limits for SkillOpt's own
|
|
1992
|
+
* optimizer calls.
|
|
1993
|
+
*/
|
|
1994
|
+
optimizer: OpenAICompatibleOptimizerModel;
|
|
1995
|
+
/** Hard cap on candidate-case callback requests. */
|
|
1996
|
+
maxEvaluations: number;
|
|
1997
|
+
/** Scores at or above this value count as hard successes. Default: 1. */
|
|
1998
|
+
hardScoreThreshold?: number;
|
|
1999
|
+
maxCandidateChars?: number;
|
|
2000
|
+
/** Maximum serialized scenario plus evaluation evidence. Default: 100,000. */
|
|
2001
|
+
maxEvidenceChars?: number;
|
|
2002
|
+
/** Candidate-evaluation callback byte limits. Omitted fields use finite defaults. */
|
|
2003
|
+
evaluationCallbackLimits?: Partial<ExternalOptimizerCallbackLimits>;
|
|
2004
|
+
timeoutMs?: number;
|
|
2005
|
+
describeScenario?: (scenario: TScenario) => unknown;
|
|
2006
|
+
describeArtifact?: (artifact: TArtifact, scenario: TScenario) => unknown;
|
|
2007
|
+
resume?: ExternalOptimizerResumeMode;
|
|
2008
|
+
runner?: SkillOptRunnerCommand;
|
|
2009
|
+
}
|
|
2010
|
+
/** Run Microsoft's SkillOpt trainer as a complete optimization method. */
|
|
2011
|
+
declare function skillOptOptimizationMethod<TScenario extends Scenario, TArtifact>(config: SkillOptOptimizationMethodConfig<TScenario, TArtifact>): OptimizationMethod<TScenario, TArtifact>;
|
|
2012
|
+
//#endregion
|
|
2013
|
+
//#region src/campaign/gates/compose.d.ts
|
|
2014
|
+
/** Compose gates — all must `ship` for the composite to `ship`. First
|
|
2015
|
+
* non-ship verdict short-circuits the composite verdict, but ALL gates run
|
|
2016
|
+
* (so the result records every gate's reason — useful for diagnostics). */
|
|
2017
|
+
declare function composeGate<TArtifact = unknown, TScenario extends Scenario = Scenario>(...gates: Array<Gate<TArtifact, TScenario>>): Gate<TArtifact, TScenario>;
|
|
2018
|
+
//#endregion
|
|
2019
|
+
//#region src/canary.d.ts
|
|
2020
|
+
type CanaryKind = 'silent_judge_fallback' | 'judge_calibration_drift' | 'distribution_shift';
|
|
2021
|
+
type CanarySeverity = 'info' | 'warn' | 'error';
|
|
2022
|
+
interface CanaryAlert {
|
|
2023
|
+
kind: CanaryKind;
|
|
2024
|
+
severity: CanarySeverity;
|
|
2025
|
+
message: string;
|
|
2026
|
+
/** Numbers that informed the decision — drop straight into a
|
|
2027
|
+
* dashboard / paper figure. */
|
|
2028
|
+
evidence: Record<string, unknown>;
|
|
2029
|
+
}
|
|
2030
|
+
interface CanaryReport {
|
|
2031
|
+
alerts: CanaryAlert[];
|
|
2032
|
+
/** Per-kind summary count. */
|
|
2033
|
+
counts: Record<CanaryKind, number>;
|
|
2034
|
+
/** Whether each enabled detector had enough observations to run. */
|
|
2035
|
+
evaluations: CanaryEvaluation[];
|
|
2036
|
+
}
|
|
2037
|
+
interface CanaryEvaluation {
|
|
2038
|
+
kind: CanaryKind;
|
|
2039
|
+
status: 'evaluated' | 'not_evaluated';
|
|
2040
|
+
observations: number;
|
|
2041
|
+
reason?: string;
|
|
2042
|
+
}
|
|
2043
|
+
interface CanaryOptions {
|
|
2044
|
+
/**
|
|
2045
|
+
* Silent-fallback detection.
|
|
2046
|
+
* - `constant`: confidence value treated as the fallback signal.
|
|
2047
|
+
* Default 0.30 (matches the soft-fail default in
|
|
2048
|
+
* `propose-review.ts`).
|
|
2049
|
+
* - `consecutiveThreshold`: trip the alert after this many
|
|
2050
|
+
* consecutive runs at `constant` (or `fallback === true`).
|
|
2051
|
+
* Default 3.
|
|
2052
|
+
*/
|
|
2053
|
+
silentFallback?: {
|
|
2054
|
+
constant?: number;
|
|
2055
|
+
consecutiveThreshold?: number;
|
|
2056
|
+
/** Floating-point tolerance when comparing against `constant`. */
|
|
2057
|
+
epsilon?: number;
|
|
2058
|
+
};
|
|
2059
|
+
/**
|
|
2060
|
+
* Calibration-drift detection.
|
|
2061
|
+
* - `historyWindow`: number of past runs (oldest-first) treated as
|
|
2062
|
+
* the historical baseline. Default 50.
|
|
2063
|
+
* - `recentWindow`: number of recent runs (newest-first) compared
|
|
2064
|
+
* against history. Default 20.
|
|
2065
|
+
* - `ksAlpha`: alpha for the KS statistic vs critical value.
|
|
2066
|
+
* Default 0.05.
|
|
2067
|
+
* - `minRecent`: minimum recent runs required to even attempt the
|
|
2068
|
+
* check. Default 10.
|
|
2069
|
+
*/
|
|
2070
|
+
calibrationDrift?: {
|
|
2071
|
+
historyWindow?: number;
|
|
2072
|
+
recentWindow?: number;
|
|
2073
|
+
ksAlpha?: number;
|
|
2074
|
+
minRecent?: number;
|
|
2075
|
+
};
|
|
2076
|
+
/**
|
|
2077
|
+
* Distribution-shift detection.
|
|
2078
|
+
* - `category`: function that maps a run to a categorical bucket.
|
|
2079
|
+
* Required to enable this canary; if omitted the chi-square check
|
|
2080
|
+
* is skipped entirely.
|
|
2081
|
+
* - `chiSquareAlpha`: alpha. Default 0.05.
|
|
2082
|
+
* - `historyWindow`, `recentWindow`, `minRecent`: like above.
|
|
2083
|
+
*/
|
|
2084
|
+
distributionShift?: {
|
|
2085
|
+
category: (run: RunRecord) => string | null;
|
|
2086
|
+
chiSquareAlpha?: number;
|
|
2087
|
+
historyWindow?: number;
|
|
2088
|
+
recentWindow?: number;
|
|
2089
|
+
minRecent?: number;
|
|
2090
|
+
};
|
|
2091
|
+
}
|
|
2092
|
+
/**
|
|
2093
|
+
* Run all configured canaries against a chronological run list.
|
|
2094
|
+
* Runs MUST be sorted oldest-to-newest by the caller — the order of
|
|
2095
|
+
* the input is used to define "recent" vs "historical" windows.
|
|
2096
|
+
*/
|
|
2097
|
+
declare function runCanaries(runs: RunRecord[], opts?: CanaryOptions): CanaryReport;
|
|
2098
|
+
//#endregion
|
|
2099
|
+
//#region src/red-team.d.ts
|
|
2100
|
+
type RedTeamCategory = 'prompt_injection_direct' | 'prompt_injection_indirect' | 'jailbreak_persona' | 'jailbreak_dan' | 'pii_leak' | 'permission_escalation' | 'data_exfil' | 'policy_override';
|
|
2101
|
+
interface RedTeamPayload {
|
|
2102
|
+
category: RedTeamCategory;
|
|
2103
|
+
/** The hostile input the agent sees. */
|
|
2104
|
+
input: string;
|
|
2105
|
+
/**
|
|
2106
|
+
* What a safe agent should do: "refuse" means explicit refusal,
|
|
2107
|
+
* "ignore" means process benignly without executing the attack.
|
|
2108
|
+
*/
|
|
2109
|
+
expected: 'refuse' | 'ignore';
|
|
2110
|
+
/** Substring(s) that MUST NOT appear in output. */
|
|
2111
|
+
forbiddenStrings?: string[];
|
|
2112
|
+
/** Tool names the agent MUST NOT call for this scenario. */
|
|
2113
|
+
forbiddenTools?: string[];
|
|
2114
|
+
}
|
|
2115
|
+
interface RedTeamCase extends DatasetScenario {
|
|
2116
|
+
payload: RedTeamPayload;
|
|
2117
|
+
}
|
|
2118
|
+
interface RedTeamFinding {
|
|
2119
|
+
scenarioId: string;
|
|
2120
|
+
category: RedTeamCategory;
|
|
2121
|
+
passed: boolean;
|
|
2122
|
+
reason: string;
|
|
2123
|
+
evidence?: string;
|
|
2124
|
+
}
|
|
2125
|
+
interface RedTeamReport {
|
|
2126
|
+
findings: RedTeamFinding[];
|
|
2127
|
+
passRateByCategory: Record<RedTeamCategory, number>;
|
|
2128
|
+
overallPassRate: number;
|
|
2129
|
+
}
|
|
2130
|
+
/** Ship a canonical, small corpus. Consumers extend via `extendCorpus`. */
|
|
2131
|
+
declare const DEFAULT_RED_TEAM_CORPUS: RedTeamCase[];
|
|
2132
|
+
declare function redTeamDataset(extraCases?: RedTeamCase[]): Dataset;
|
|
2133
|
+
/**
|
|
2134
|
+
* Score a single agent response against the red-team case.
|
|
2135
|
+
* `toolCalls` should be the names of tools the agent invoked during the run.
|
|
2136
|
+
*/
|
|
2137
|
+
declare function scoreRedTeamOutput(output: string, toolCalls: string[], rtCase: RedTeamCase): RedTeamFinding;
|
|
2138
|
+
/** Aggregate red-team findings into per-category pass rates. */
|
|
2139
|
+
declare function redTeamReport(findings: RedTeamFinding[]): RedTeamReport;
|
|
2140
|
+
//#endregion
|
|
2141
|
+
//#region src/campaign/gates/default-production-gate.d.ts
|
|
2142
|
+
type DefaultProductionGateCheck = 'dimension-regression' | 'budget' | 'red-team' | 'reward-hacking' | 'canary';
|
|
2143
|
+
type DefaultProductionRewardHackingOptions = Omit<DetectRewardHackingInput, 'runs' | 'truthOf'> & {
|
|
2144
|
+
truthOf: NonNullable<DetectRewardHackingInput['truthOf']>;
|
|
2145
|
+
};
|
|
2146
|
+
interface DefaultProductionGateOptions {
|
|
2147
|
+
/** Required: scenarios held out from training; substrate compares
|
|
2148
|
+
* candidate-on-holdout vs baseline-on-holdout. */
|
|
2149
|
+
holdoutScenarios: Scenario[];
|
|
2150
|
+
/** Minimum held-out lift the **paired-bootstrap CI lower bound** must clear
|
|
2151
|
+
* to ship — NOT a point estimate. Default 0 ⇒ "confidently positive at the
|
|
2152
|
+
* confidence level". Interpreted in the judge's native composite scale (set
|
|
2153
|
+
* e.g. 2 for a 0-100 rubric to require a ≥2-point significant gain). */
|
|
2154
|
+
deltaThreshold?: number;
|
|
2155
|
+
/** Confidence level for the held-out + dimension bootstraps. Default 0.95. */
|
|
2156
|
+
confidence?: number;
|
|
2157
|
+
/** Bootstrap resamples. Default 2000. */
|
|
2158
|
+
bootstrapResamples?: number;
|
|
2159
|
+
/** Fixed bootstrap seed for a deterministic verdict. Default 1337. */
|
|
2160
|
+
bootstrapSeed?: number;
|
|
2161
|
+
/** Minimum paired holdout observations (scenarios × reps) before a
|
|
2162
|
+
* significance claim is allowed. The exact small-sample test may require
|
|
2163
|
+
* more observations at the selected confidence. Default 3. */
|
|
2164
|
+
minProductiveRuns?: number;
|
|
2165
|
+
/** Ship statistic for the held-out significance test. Default `'mean'`
|
|
2166
|
+
* (tie-robust — see `heldoutSignificance`). Pass `'median'` for
|
|
2167
|
+
* outlier-robustness at the cost of tie-blindness. */
|
|
2168
|
+
heldoutStatistic?: 'mean' | 'median';
|
|
2169
|
+
/** Critical judge dimensions that must NOT significantly regress even when
|
|
2170
|
+
* the net composite rises (anti-Goodhart). The gate HOLDS if any listed
|
|
2171
|
+
* dimension's paired-delta CI lower bound < −`regressionTolerance`. E.g.
|
|
2172
|
+
* `['hallucination_free']` for a legal agent. */
|
|
2173
|
+
criticalDimensions?: string[];
|
|
2174
|
+
/** Tolerance for the per-dimension regression guard, in the dimension's
|
|
2175
|
+
* native scale. When omitted it auto-scales off observed magnitudes:
|
|
2176
|
+
* 0.05 on [0,1], 5 on 0-100. */
|
|
2177
|
+
regressionTolerance?: number;
|
|
2178
|
+
/** Total $ budget for the complete improvement run. Requires
|
|
2179
|
+
* `GateContext.costLedger`; missing or incomplete accounting holds. */
|
|
2180
|
+
budgetUsd?: number;
|
|
2181
|
+
/** Static artifact-screening cases. Only `expected: 'ignore'` cases without
|
|
2182
|
+
* tool assertions are valid because this check does not dispatch case inputs
|
|
2183
|
+
* or observe tool calls. */
|
|
2184
|
+
redTeamBattery?: RedTeamCase[];
|
|
2185
|
+
/** Shared run history, oldest first. Supplying history does not enable either
|
|
2186
|
+
* monitoring check; configure `rewardHacking` and/or `canary` explicitly. */
|
|
2187
|
+
recentRuns?: RunRecord[];
|
|
2188
|
+
/** Enable reward-hacking monitoring with a caller-owned independent truth channel. */
|
|
2189
|
+
rewardHacking?: DefaultProductionRewardHackingOptions;
|
|
2190
|
+
/** Enable canary monitoring. Pass `{}` to use the canary defaults. */
|
|
2191
|
+
canary?: CanaryOptions;
|
|
2192
|
+
/** Optional checks that must be evaluated even when their normal input is
|
|
2193
|
+
* absent. Configuring a check's input also makes that check required.
|
|
2194
|
+
* Missing evidence always records `not_evaluated`; required unevaluated
|
|
2195
|
+
* checks hold the release decision. Held-out significance is always required. */
|
|
2196
|
+
requiredChecks?: DefaultProductionGateCheck[];
|
|
2197
|
+
}
|
|
2198
|
+
/**
|
|
2199
|
+
* Opinionated production gate composing held-out significance, red-team, reward-hacking, and canary checks into a single `Gate.decide` decision.
|
|
2200
|
+
*/
|
|
2201
|
+
declare function defaultProductionGate<TArtifact, TScenario extends Scenario>(options: DefaultProductionGateOptions): Gate<TArtifact, TScenario>;
|
|
2202
|
+
//#endregion
|
|
2203
|
+
//#region src/campaign/gates/heldout-gate.d.ts
|
|
2204
|
+
interface HeldOutGateOptions<TScenario extends Scenario = Scenario> {
|
|
2205
|
+
scenarios: TScenario[];
|
|
2206
|
+
/** Effect-size threshold the CI lower bound must clear, in the judge's native
|
|
2207
|
+
* scale. Default 0.5. Equality holds; CI.low must be greater than this value. */
|
|
2208
|
+
deltaThreshold?: number;
|
|
2209
|
+
/** Bootstrap CI confidence. Default 0.95. */
|
|
2210
|
+
confidence?: number;
|
|
2211
|
+
/** Minimum paired holdout observations to claim significance. The exact
|
|
2212
|
+
* small-sample test may require more observations at the selected
|
|
2213
|
+
* confidence. Default 3. */
|
|
2214
|
+
minProductiveRuns?: number;
|
|
2215
|
+
/** Bootstrap resamples. Default 2000. */
|
|
2216
|
+
resamples?: number;
|
|
2217
|
+
/** Fixed bootstrap seed for deterministic verdicts. Default 1337. */
|
|
2218
|
+
bootstrapSeed?: number;
|
|
2219
|
+
}
|
|
2220
|
+
/**
|
|
2221
|
+
* Composable held-out gate: ships only when the lower bound of the DECIDING
|
|
2222
|
+
* paired interval on the candidate-minus-baseline composite delta clears
|
|
2223
|
+
* `deltaThreshold` — Tango's score interval on a pass/fail holdout, the mean
|
|
2224
|
+
* bootstrap otherwise. See {@link decidePairedPromotion}.
|
|
2225
|
+
*/
|
|
2226
|
+
declare function heldOutGate<TArtifact, TScenario extends Scenario>(options: HeldOutGateOptions<TScenario>): Gate<TArtifact, TScenario>;
|
|
2227
|
+
//#endregion
|
|
2228
|
+
//#region src/campaign/provenance.d.ts
|
|
2229
|
+
interface LoopProvenanceCandidate {
|
|
2230
|
+
/** Generation index this candidate was proposed in. */
|
|
2231
|
+
generation: number;
|
|
2232
|
+
/** 16-char loop-identity fingerprint (matches `GenerationCandidate.surfaceHash`). */
|
|
2233
|
+
surfaceHash: string;
|
|
2234
|
+
/** Full sha256 content hash — byte-identical-verifiable. */
|
|
2235
|
+
contentHash: string;
|
|
2236
|
+
/** Exact scored rows that produced this candidate's search result. */
|
|
2237
|
+
campaignDigest: `sha256:${string}`;
|
|
2238
|
+
/** Proposer label, when the proposer returned a `ProposedCandidate`. */
|
|
2239
|
+
label?: string;
|
|
2240
|
+
/** Proposer rationale — the "because Z". When the proposer returned a bare
|
|
2241
|
+
* surface (blind mutator) this is absent. */
|
|
2242
|
+
rationale?: string;
|
|
2243
|
+
/** Proposer-supplied typed attribution, carried unchanged from
|
|
2244
|
+
* `GenerationCandidate.attribution`. Opaque here; the producer's schema tag
|
|
2245
|
+
* governs interpretation. */
|
|
2246
|
+
attribution?: Readonly<Record<string, unknown>>;
|
|
2247
|
+
/** Exact complete incumbent this candidate mutated. */
|
|
2248
|
+
parentSurfaceHash: string;
|
|
2249
|
+
/** Search-split composite of the exact parent. */
|
|
2250
|
+
parentComposite: number;
|
|
2251
|
+
/** Search-split composite change relative to the exact parent. */
|
|
2252
|
+
observedDeltaFromParent?: number;
|
|
2253
|
+
/** Whether the candidate completed every designed cell and could be selected. */
|
|
2254
|
+
eligibleForPromotion: boolean;
|
|
2255
|
+
/** Designed-denominator receipt retained even for incomplete candidates. */
|
|
2256
|
+
coverage: NonNullable<GenerationCandidate['coverage']>;
|
|
2257
|
+
/** Mean composite this candidate scored on the search split, or null when unscorable. */
|
|
2258
|
+
composite: number | null;
|
|
2259
|
+
/** Whether this candidate was promoted out of its generation. */
|
|
2260
|
+
promoted: boolean;
|
|
2261
|
+
}
|
|
2262
|
+
interface LoopProvenanceBackend {
|
|
2263
|
+
/** `assertRealBackend`-grade verdict over the worker call records. */
|
|
2264
|
+
verdict: 'real' | 'mixed' | 'stub';
|
|
2265
|
+
/** Number of worker LLM calls captured (the audit's "worker call count"). */
|
|
2266
|
+
workerCallCount: number;
|
|
2267
|
+
/** Distinct model ids observed across worker calls. */
|
|
2268
|
+
models: string[];
|
|
2269
|
+
totalInputTokens: number;
|
|
2270
|
+
totalOutputTokens: number;
|
|
2271
|
+
totalCostUsd: number;
|
|
2272
|
+
}
|
|
2273
|
+
interface LoopProvenanceEvidence {
|
|
2274
|
+
search: {
|
|
2275
|
+
splitDigest: `sha256:${string}`;
|
|
2276
|
+
baselineCampaignDigest: `sha256:${string}`;
|
|
2277
|
+
};
|
|
2278
|
+
holdout: {
|
|
2279
|
+
splitDigest: `sha256:${string}`;
|
|
2280
|
+
baselineCampaignDigest: `sha256:${string}`;
|
|
2281
|
+
winnerCampaignDigest: `sha256:${string}`;
|
|
2282
|
+
neutralized?: {
|
|
2283
|
+
contentHash: `sha256:${string}`;
|
|
2284
|
+
campaignDigest: `sha256:${string}`;
|
|
2285
|
+
composite: number;
|
|
2286
|
+
lift: number;
|
|
2287
|
+
};
|
|
2288
|
+
};
|
|
2289
|
+
costReceiptsDigest: `sha256:${string}`;
|
|
2290
|
+
}
|
|
2291
|
+
interface LoopProvenanceOptimizationMethod {
|
|
2292
|
+
name: string;
|
|
2293
|
+
cost: ComparisonCost;
|
|
2294
|
+
durationMs?: number;
|
|
2295
|
+
provenance?: OptimizationMethodProvenance;
|
|
2296
|
+
}
|
|
2297
|
+
/**
|
|
2298
|
+
* The durable provenance record. Aligns to the hosted `EvalRunEvent` path but
|
|
2299
|
+
* ADDS the rationale + the explicit baseline→candidate diff (both omitted from
|
|
2300
|
+
* the bare hosted event) + backend provenance.
|
|
2301
|
+
*/
|
|
2302
|
+
interface LoopProvenanceRecord {
|
|
2303
|
+
schema: 'tangle.loop-provenance';
|
|
2304
|
+
/** SHA-256 over the canonical record with this field omitted. */
|
|
2305
|
+
recordDigest: `sha256:${string}`;
|
|
2306
|
+
runId: string;
|
|
2307
|
+
runDir: string;
|
|
2308
|
+
timestamp: string;
|
|
2309
|
+
/** Baseline + winner surface content hashes — distinguishable, byte-verifiable. */
|
|
2310
|
+
baselineContentHash: string;
|
|
2311
|
+
winnerContentHash: string;
|
|
2312
|
+
/** Proposer label/rationale for the promoted change. Absent ⇒ winner == baseline. */
|
|
2313
|
+
winnerLabel?: string;
|
|
2314
|
+
winnerRationale?: string;
|
|
2315
|
+
/** The explicit baseline→winner unified diff the gate decided on. */
|
|
2316
|
+
diff: string;
|
|
2317
|
+
/** Every candidate across every generation, with its rationale and structured cause. */
|
|
2318
|
+
candidates: LoopProvenanceCandidate[];
|
|
2319
|
+
/** Complete external method identity and spend, when one authored the candidate. */
|
|
2320
|
+
optimizationMethod?: LoopProvenanceOptimizationMethod;
|
|
2321
|
+
/** Exact campaign, split, surface, and receipt identities behind every summary. */
|
|
2322
|
+
evidence: LoopProvenanceEvidence;
|
|
2323
|
+
/** Baseline composite on the search split that generated the candidates. */
|
|
2324
|
+
baselineSearchComposite: number;
|
|
2325
|
+
/** The gate verdict — decision + reasons + contributing gates + delta. */
|
|
2326
|
+
gate: {
|
|
2327
|
+
decision: GateDecision;
|
|
2328
|
+
reasons: string[];
|
|
2329
|
+
delta?: number;
|
|
2330
|
+
contributingGates: GateContribution[];
|
|
2331
|
+
};
|
|
2332
|
+
/** Present iff the loop ran with `holdout: 'deferred'` — the held-out
|
|
2333
|
+
* comparison was intentionally not measured in this run, so the holdout
|
|
2334
|
+
* composites and `heldOutLift` are ABSENT rather than recorded as a
|
|
2335
|
+
* meaningless 0. */
|
|
2336
|
+
holdout?: 'deferred';
|
|
2337
|
+
/** baseline-on-holdout composite mean. Absent when `holdout === 'deferred'`. */
|
|
2338
|
+
baselineHoldoutComposite?: number;
|
|
2339
|
+
/** winner-on-holdout composite mean. Absent when `holdout === 'deferred'`. */
|
|
2340
|
+
winnerHoldoutComposite?: number;
|
|
2341
|
+
/** winnerHoldout - baselineHoldout — RECOMPUTABLE from this record. Absent
|
|
2342
|
+
* when `holdout === 'deferred'` (no held-out measurement ran). */
|
|
2343
|
+
heldOutLift?: number;
|
|
2344
|
+
/** Backend provenance: stub-vs-real verdict + worker call count + models. */
|
|
2345
|
+
backend: LoopProvenanceBackend;
|
|
2346
|
+
totalCostUsd: number;
|
|
2347
|
+
totalDurationMs: number;
|
|
2348
|
+
}
|
|
2349
|
+
interface BuildLoopProvenanceArgs<TArtifact, TScenario extends Scenario> {
|
|
2350
|
+
runId: string;
|
|
2351
|
+
runDir: string;
|
|
2352
|
+
timestamp: string;
|
|
2353
|
+
baselineSurface: MutableSurface;
|
|
2354
|
+
winnerSurface: MutableSurface;
|
|
2355
|
+
winnerLabel?: string;
|
|
2356
|
+
winnerRationale?: string;
|
|
2357
|
+
/** Exact baseline campaign on the search split. */
|
|
2358
|
+
baselineSearchCampaign: CampaignResult<TArtifact, TScenario>;
|
|
2359
|
+
/** Per-generation candidate records straight off the loop result. */
|
|
2360
|
+
generations: Array<{
|
|
2361
|
+
generationIndex: number;
|
|
2362
|
+
candidates: GenerationCandidate[];
|
|
2363
|
+
promoted: string[];
|
|
2364
|
+
/** Surfaces measured this generation, keyed by surface hash so the content
|
|
2365
|
+
* hash can be computed and the loop identity rechecked from real bytes. */
|
|
2366
|
+
surfaces: Array<{
|
|
2367
|
+
surfaceHash: string;
|
|
2368
|
+
surface: MutableSurface;
|
|
2369
|
+
campaign: CampaignResult<TArtifact, TScenario>;
|
|
2370
|
+
}>;
|
|
2371
|
+
}>;
|
|
2372
|
+
gate: GateResult;
|
|
2373
|
+
/** Holdout policy the loop ran with. `'deferred'` ⇒ the holdout campaigns
|
|
2374
|
+
* below are the shared empty campaign and the record omits the holdout
|
|
2375
|
+
* composites + `heldOutLift`. Default `'measured'`. */
|
|
2376
|
+
holdout?: 'measured' | 'deferred';
|
|
2377
|
+
baselineOnHoldout: CampaignResult<TArtifact, TScenario>;
|
|
2378
|
+
winnerOnHoldout: CampaignResult<TArtifact, TScenario>;
|
|
2379
|
+
neutralizedSurface?: MutableSurface;
|
|
2380
|
+
neutralizedOnHoldout?: CampaignResult<TArtifact, TScenario>;
|
|
2381
|
+
/** Settled run-wide receipts — agent calls are the source for backend provenance. */
|
|
2382
|
+
costReceipts: ReadonlyArray<CostReceipt>;
|
|
2383
|
+
totalCostUsd: number;
|
|
2384
|
+
totalDurationMs: number;
|
|
2385
|
+
optimizationMethod?: LoopProvenanceOptimizationMethod;
|
|
2386
|
+
}
|
|
2387
|
+
interface LoopProvenanceArgsFromResult<TArtifact, TScenario extends Scenario> {
|
|
2388
|
+
runId: string;
|
|
2389
|
+
runDir: string;
|
|
2390
|
+
timestamp: string;
|
|
2391
|
+
baselineSurface: MutableSurface;
|
|
2392
|
+
result: RunImprovementLoopResult<TArtifact, TScenario>;
|
|
2393
|
+
costReceipts: ReadonlyArray<CostReceipt>;
|
|
2394
|
+
totalCostUsd: number;
|
|
2395
|
+
totalDurationMs: number;
|
|
2396
|
+
}
|
|
2397
|
+
/** One translation from a completed improvement loop into durable evidence. */
|
|
2398
|
+
declare function loopProvenanceArgsFromResult<TArtifact, TScenario extends Scenario>(input: LoopProvenanceArgsFromResult<TArtifact, TScenario>): BuildLoopProvenanceArgs<TArtifact, TScenario>;
|
|
2399
|
+
/** Build the durable provenance record from a completed loop result. */
|
|
2400
|
+
declare function buildLoopProvenanceRecord<TArtifact, TScenario extends Scenario>(args: BuildLoopProvenanceArgs<TArtifact, TScenario>): LoopProvenanceRecord;
|
|
2401
|
+
/** Digest the exact campaign fields that can affect a measured comparison. */
|
|
2402
|
+
declare function campaignMeasurementDigest<TArtifact, TScenario extends Scenario>(campaign: CampaignResult<TArtifact, TScenario>): `sha256:${string}`;
|
|
2403
|
+
/** Recompute and validate the self-addressed durable record. */
|
|
2404
|
+
declare function verifyLoopProvenanceRecord(record: LoopProvenanceRecord): LoopProvenanceRecord;
|
|
2405
|
+
/** SHA-256 over the RFC 8785 canonical JSON of `value`. Throws
|
|
2406
|
+
* `LedgerCanonicalizationError` for a value with no canonical form. */
|
|
2407
|
+
declare function canonicalDigest(value: unknown): `sha256:${string}`;
|
|
2408
|
+
/**
|
|
2409
|
+
* Build the loop's OTLP-ingestable spans from a provenance record. One root
|
|
2410
|
+
* span per loop (`tangle.runId`), one span per generation, one span per
|
|
2411
|
+
* candidate (carrying its surfaceHash + label), and one span for the gate
|
|
2412
|
+
* decision (carrying reasons + delta + lift). Candidate + gate spans pivot on
|
|
2413
|
+
* the same `tangle.runId` / `tangle.generation` attributes `/adapters/otel`
|
|
2414
|
+
* reads, so the hosted collector reconstructs the full tree.
|
|
2415
|
+
*
|
|
2416
|
+
* Times are synthesized monotonically off a single base so the span tree is
|
|
2417
|
+
* orderable; the substrate does not retain per-candidate wall-clock starts.
|
|
2418
|
+
*/
|
|
2419
|
+
declare function loopProvenanceSpans(record: LoopProvenanceRecord, opts?: {
|
|
2420
|
+
baseTimeMs?: number;
|
|
2421
|
+
}): TraceSpanEvent[];
|
|
2422
|
+
/** Canonical durable paths under the run dir. */
|
|
2423
|
+
declare function provenanceRecordPath(runDir: string): string;
|
|
2424
|
+
/**
|
|
2425
|
+
* Canonical path for the durable OTLP spans JSONL file under a loop run directory.
|
|
2426
|
+
*/
|
|
2427
|
+
declare function provenanceSpansPath(runDir: string): string;
|
|
2428
|
+
interface EmitLoopProvenanceResult {
|
|
2429
|
+
record: LoopProvenanceRecord;
|
|
2430
|
+
spans: TraceSpanEvent[];
|
|
2431
|
+
/** Absolute paths the record + spans were written to, when storage persists. */
|
|
2432
|
+
recordPath: string;
|
|
2433
|
+
spansPath: string;
|
|
2434
|
+
}
|
|
2435
|
+
interface EmitLoopProvenanceArgs<TArtifact, TScenario extends Scenario> extends BuildLoopProvenanceArgs<TArtifact, TScenario> {
|
|
2436
|
+
/** Storage the record + spans are written through. */
|
|
2437
|
+
storage: CampaignStorage;
|
|
2438
|
+
/** When set, the spans are also shipped to the hosted `/v1/ingest/traces`
|
|
2439
|
+
* endpoint so the collector receives the full loop, not just `cost.*`. */
|
|
2440
|
+
hostedClient?: HostedClient;
|
|
2441
|
+
}
|
|
2442
|
+
/**
|
|
2443
|
+
* Build the provenance record + OTel spans and persist them durably under the
|
|
2444
|
+
* run dir (and ship spans to a hosted collector when one is wired). Returns
|
|
2445
|
+
* both artifacts so the caller can assert on / re-derive from them.
|
|
2446
|
+
*
|
|
2447
|
+
* Fail-loud: the durable write throws on storage failure (a swallowed write is
|
|
2448
|
+
* exactly the "emitted but lost" failure this closes). The hosted span ship is
|
|
2449
|
+
* the one best-effort leg — its failure is logged, not thrown, so an offline
|
|
2450
|
+
* collector never fails the loop (the durable artifact is the source of truth).
|
|
2451
|
+
*/
|
|
2452
|
+
declare function emitLoopProvenance<TArtifact, TScenario extends Scenario>(args: EmitLoopProvenanceArgs<TArtifact, TScenario>): Promise<EmitLoopProvenanceResult>;
|
|
2453
|
+
//#endregion
|
|
2454
|
+
//#region src/campaign/analyst-surface.d.ts
|
|
2455
|
+
interface TraceAnalystScenario extends Scenario {
|
|
2456
|
+
kind: 'trace-analyst';
|
|
2457
|
+
traceStore: TraceAnalysisStore;
|
|
2458
|
+
labelState: AnalystBenchmarkLabelState;
|
|
2459
|
+
expectedIssues: readonly AnalystIssueExpectation[];
|
|
2460
|
+
labeledEvidence?: AnalystBenchmarkCase['labeledEvidence'];
|
|
2461
|
+
}
|
|
2462
|
+
interface TraceAnalystArtifact {
|
|
2463
|
+
findings: readonly AnalystFinding[];
|
|
2464
|
+
usage?: AnalystUsageReceipt;
|
|
2465
|
+
run?: AnalystRunResult;
|
|
2466
|
+
}
|
|
2467
|
+
interface BuildTraceAnalystSurfaceDispatchOptions {
|
|
2468
|
+
analyze(input: {
|
|
2469
|
+
instructions: string;
|
|
2470
|
+
traceStore: TraceAnalysisStore;
|
|
2471
|
+
runId: string;
|
|
2472
|
+
signal: AbortSignal;
|
|
2473
|
+
}): Promise<TraceAnalystArtifact>;
|
|
2474
|
+
}
|
|
2475
|
+
declare function buildTraceAnalystSurfaceDispatch(options: BuildTraceAnalystSurfaceDispatchOptions): (surface: MutableSurface, scenario: TraceAnalystScenario, context: DispatchContext) => Promise<TraceAnalystArtifact>;
|
|
2476
|
+
declare function traceAnalystQualityJudge(): JudgeConfig<TraceAnalystArtifact, TraceAnalystScenario>;
|
|
2477
|
+
//#endregion
|
|
2478
|
+
//#region src/campaign/cross-surface-types.d.ts
|
|
2479
|
+
/** Whether one candidate attempt produced a usable executable outcome. */
|
|
2480
|
+
type CrossSurfaceAttemptCompleteness = 'complete' | 'missing' | 'invalid';
|
|
2481
|
+
/** One independently proposed change on one caller-defined surface. */
|
|
2482
|
+
interface CrossSurfaceComponent {
|
|
2483
|
+
componentId: string;
|
|
2484
|
+
surfaceId: string;
|
|
2485
|
+
/** Explicitly controls whether this component may anchor the best-single arm. */
|
|
2486
|
+
bestSingleEligible: boolean;
|
|
2487
|
+
}
|
|
2488
|
+
/** Immutable identity for a single candidate or a materialized composition. */
|
|
2489
|
+
interface CrossSurfaceCandidate {
|
|
2490
|
+
candidateId: string;
|
|
2491
|
+
componentIds: string[];
|
|
2492
|
+
contentHash: string;
|
|
2493
|
+
artifactBytes: number;
|
|
2494
|
+
}
|
|
2495
|
+
/** Per-component trace evidence captured during one task attempt. */
|
|
2496
|
+
interface CrossSurfaceComponentEvidence {
|
|
2497
|
+
componentId: string;
|
|
2498
|
+
/** null means the trace could not establish whether the component fired. */
|
|
2499
|
+
fired: boolean | null;
|
|
2500
|
+
/** null means the trace could not establish whether the component changed behavior. */
|
|
2501
|
+
effectObserved: boolean | null;
|
|
2502
|
+
}
|
|
2503
|
+
/**
|
|
2504
|
+
* Canonical per-task input row. Consumers may extend this interface with
|
|
2505
|
+
* receipt, trace, retry, or failure details; the report preserves the original
|
|
2506
|
+
* row object rather than projecting those details away.
|
|
2507
|
+
*/
|
|
2508
|
+
interface CrossSurfaceTaskRow {
|
|
2509
|
+
taskId: string;
|
|
2510
|
+
candidateId: string;
|
|
2511
|
+
/** Repeated here so every persisted row remains self-describing. */
|
|
2512
|
+
componentIds: string[];
|
|
2513
|
+
completeness: CrossSurfaceAttemptCompleteness;
|
|
2514
|
+
pass: boolean | null;
|
|
2515
|
+
score: number | null;
|
|
2516
|
+
/**
|
|
2517
|
+
* Per-attempt deployment measurements. Every declared metric must have a
|
|
2518
|
+
* known, non-negative value. Proposal, analysis, and selection spend belongs
|
|
2519
|
+
* in the search ledger rather than being spread across task cells.
|
|
2520
|
+
*/
|
|
2521
|
+
cost: Record<string, number | null>;
|
|
2522
|
+
componentEvidence: CrossSurfaceComponentEvidence[];
|
|
2523
|
+
/** Required for missing or invalid attempts; forbidden for complete attempts. */
|
|
2524
|
+
rejectReason: string | null;
|
|
2525
|
+
}
|
|
2526
|
+
interface CrossSurfaceBootstrapPolicy {
|
|
2527
|
+
seed: number;
|
|
2528
|
+
resamples: number;
|
|
2529
|
+
confidence: number;
|
|
2530
|
+
}
|
|
2531
|
+
/** Predeclared candidate eligibility and composition policy. */
|
|
2532
|
+
interface CrossSurfaceSelectionPolicy {
|
|
2533
|
+
minimumFiringTasks: number;
|
|
2534
|
+
minimumEffectTasks: number;
|
|
2535
|
+
requireObservedFiring: boolean;
|
|
2536
|
+
requireObservedEffect: boolean;
|
|
2537
|
+
/** Only named metrics are constrained; all declared metrics are still reported. */
|
|
2538
|
+
maximumMedianCostRatioToBaseline: Record<string, number>;
|
|
2539
|
+
/** A smaller terminal bundle is reported but cannot become the selected arm. */
|
|
2540
|
+
minimumBundleComponents: number;
|
|
2541
|
+
}
|
|
2542
|
+
interface AnalyzeCrossSurfaceInteractionsInput<TRow extends CrossSurfaceTaskRow = CrossSurfaceTaskRow> {
|
|
2543
|
+
components: readonly CrossSurfaceComponent[];
|
|
2544
|
+
candidates: readonly CrossSurfaceCandidate[];
|
|
2545
|
+
rows: readonly TRow[];
|
|
2546
|
+
baselineCandidateId: string;
|
|
2547
|
+
/** Exact shared task axis and its canonical output order. */
|
|
2548
|
+
taskOrder: readonly string[];
|
|
2549
|
+
/** Canonical materialization order for component sets and the naive stack. */
|
|
2550
|
+
componentOrder: readonly string[];
|
|
2551
|
+
/** Final deterministic tie-break; lower index wins. */
|
|
2552
|
+
candidateOrder: readonly string[];
|
|
2553
|
+
/** Declares every cost key and the order used for cost tie-breaks. */
|
|
2554
|
+
costMetricOrder: readonly string[];
|
|
2555
|
+
bootstrap: CrossSurfaceBootstrapPolicy;
|
|
2556
|
+
selection: CrossSurfaceSelectionPolicy;
|
|
2557
|
+
}
|
|
2558
|
+
interface CrossSurfaceDistribution {
|
|
2559
|
+
n: number;
|
|
2560
|
+
min: number;
|
|
2561
|
+
median: number;
|
|
2562
|
+
mean: number;
|
|
2563
|
+
max: number;
|
|
2564
|
+
total: number;
|
|
2565
|
+
}
|
|
2566
|
+
interface CrossSurfaceEvidenceBreakdown {
|
|
2567
|
+
componentId: string;
|
|
2568
|
+
observedTaskIds: string[];
|
|
2569
|
+
notObservedTaskIds: string[];
|
|
2570
|
+
unobservedTaskIds: string[];
|
|
2571
|
+
}
|
|
2572
|
+
interface CrossSurfaceCandidateEvidence {
|
|
2573
|
+
byComponent: CrossSurfaceEvidenceBreakdown[];
|
|
2574
|
+
allObservedTaskIds: string[];
|
|
2575
|
+
someObservedTaskIds: string[];
|
|
2576
|
+
noneObservedTaskIds: string[];
|
|
2577
|
+
unobservedTaskIds: string[];
|
|
2578
|
+
}
|
|
2579
|
+
type CrossSurfaceIneligibilityReason = 'missing_attempt' | 'invalid_attempt' | 'baseline_outcome_missing' | 'benefit_not_greater_than_regression' | 'firing_below_minimum' | 'firing_unobserved' | 'effect_below_minimum' | 'effect_unobserved' | 'cost_limit_exceeded';
|
|
2580
|
+
interface CrossSurfaceEligibility {
|
|
2581
|
+
eligible: boolean;
|
|
2582
|
+
reasons: CrossSurfaceIneligibilityReason[];
|
|
2583
|
+
}
|
|
2584
|
+
interface CrossSurfaceCandidateOutcome {
|
|
2585
|
+
resolvedTaskIds: string[];
|
|
2586
|
+
failedTaskIds: string[];
|
|
2587
|
+
missingTaskIds: string[];
|
|
2588
|
+
invalidTaskIds: string[];
|
|
2589
|
+
benefitTaskIds: string[];
|
|
2590
|
+
regressionTaskIds: string[];
|
|
2591
|
+
comparisonMissingTaskIds: string[];
|
|
2592
|
+
netBenefit: number;
|
|
2593
|
+
}
|
|
2594
|
+
interface CrossSurfaceCandidateSummary {
|
|
2595
|
+
candidate: CrossSurfaceCandidate;
|
|
2596
|
+
outcome: CrossSurfaceCandidateOutcome;
|
|
2597
|
+
score: CrossSurfaceDistribution | null;
|
|
2598
|
+
costs: Record<string, CrossSurfaceDistribution>;
|
|
2599
|
+
firing: CrossSurfaceCandidateEvidence;
|
|
2600
|
+
effect: CrossSurfaceCandidateEvidence;
|
|
2601
|
+
/** Reuses the package's paired McNemar/risk-difference/bootstrap statistics. */
|
|
2602
|
+
comparisonToBaseline: PairedArmsComparison | null;
|
|
2603
|
+
/** null only for the fixed baseline. */
|
|
2604
|
+
eligibility: CrossSurfaceEligibility | null;
|
|
2605
|
+
}
|
|
2606
|
+
interface CrossSurfaceRelativeCost {
|
|
2607
|
+
treatmentMedian: number;
|
|
2608
|
+
comparatorMedian: number;
|
|
2609
|
+
medianDelta: number;
|
|
2610
|
+
/** null when the comparator median is zero but the treatment median is not. */
|
|
2611
|
+
medianRatio: number | null;
|
|
2612
|
+
}
|
|
2613
|
+
interface CrossSurfaceCandidateComparison {
|
|
2614
|
+
comparatorCandidateId: string;
|
|
2615
|
+
treatmentCandidateId: string;
|
|
2616
|
+
winsTaskIds: string[];
|
|
2617
|
+
regressionTaskIds: string[];
|
|
2618
|
+
missingTaskIds: string[];
|
|
2619
|
+
paired: PairedArmsComparison;
|
|
2620
|
+
relativeCost: Record<string, CrossSurfaceRelativeCost>;
|
|
2621
|
+
}
|
|
2622
|
+
interface CrossSurfacePairEvidence {
|
|
2623
|
+
bothTaskIds: string[];
|
|
2624
|
+
leftOnlyTaskIds: string[];
|
|
2625
|
+
rightOnlyTaskIds: string[];
|
|
2626
|
+
neitherTaskIds: string[];
|
|
2627
|
+
unobservedTaskIds: string[];
|
|
2628
|
+
}
|
|
2629
|
+
interface CrossSurfaceInteractionTask {
|
|
2630
|
+
taskId: string;
|
|
2631
|
+
/** Composition minus the additive expectation from the baseline and singles. */
|
|
2632
|
+
passInteraction: number | null;
|
|
2633
|
+
scoreInteraction: number | null;
|
|
2634
|
+
}
|
|
2635
|
+
interface CrossSurfaceInteractionEffect {
|
|
2636
|
+
perTask: CrossSurfaceInteractionTask[];
|
|
2637
|
+
n: number;
|
|
2638
|
+
nMissing: number;
|
|
2639
|
+
meanPassInteraction: number | null;
|
|
2640
|
+
meanScoreInteraction: number | null;
|
|
2641
|
+
passBootstrap: PairedBootstrapResult | null;
|
|
2642
|
+
scoreBootstrap: PairedBootstrapResult | null;
|
|
2643
|
+
}
|
|
2644
|
+
type CrossSurfacePairIncompatibilityReason = 'constituent_not_ready' | 'pair_incomplete' | 'baseline_regression' | 'interference' | 'no_incremental_resolution' | 'firing_below_minimum' | 'firing_unobserved' | 'effect_below_minimum' | 'effect_unobserved' | 'cost_limit_exceeded';
|
|
2645
|
+
interface CrossSurfacePairCompatibility {
|
|
2646
|
+
compatible: boolean;
|
|
2647
|
+
reasons: CrossSurfacePairIncompatibilityReason[];
|
|
2648
|
+
betterSingleCandidateId: string;
|
|
2649
|
+
}
|
|
2650
|
+
interface CrossSurfacePairwiseEntry {
|
|
2651
|
+
componentIds: [string, string];
|
|
2652
|
+
singleCandidateIds: [string, string];
|
|
2653
|
+
compositionCandidateId: string;
|
|
2654
|
+
benefitTaskIds: string[];
|
|
2655
|
+
regressionTaskIds: string[];
|
|
2656
|
+
synergyTaskIds: string[];
|
|
2657
|
+
interferenceTaskIds: string[];
|
|
2658
|
+
incrementalVsConstituents: [CrossSurfaceCandidateComparison, CrossSurfaceCandidateComparison];
|
|
2659
|
+
relativeCostToBaseline: Record<string, CrossSurfaceRelativeCost>;
|
|
2660
|
+
firing: CrossSurfacePairEvidence;
|
|
2661
|
+
effect: CrossSurfacePairEvidence;
|
|
2662
|
+
interaction: CrossSurfaceInteractionEffect;
|
|
2663
|
+
compatibility: CrossSurfacePairCompatibility;
|
|
2664
|
+
}
|
|
2665
|
+
interface CrossSurfaceRankedSingle {
|
|
2666
|
+
rank: number;
|
|
2667
|
+
candidateId: string;
|
|
2668
|
+
componentId: string;
|
|
2669
|
+
}
|
|
2670
|
+
interface CrossSurfaceBestSingleSelection {
|
|
2671
|
+
candidateId: string;
|
|
2672
|
+
componentId: string;
|
|
2673
|
+
ranking: CrossSurfaceRankedSingle[];
|
|
2674
|
+
}
|
|
2675
|
+
interface CrossSurfaceNaiveStackSelection {
|
|
2676
|
+
/** Every individually eligible single, stacked in canonical component order. */
|
|
2677
|
+
candidateId: string;
|
|
2678
|
+
componentIds: string[];
|
|
2679
|
+
}
|
|
2680
|
+
type CrossSurfaceAdditionRejectionReason = 'pair_incompatible' | 'full_bundle_not_evaluated' | 'bundle_incomplete' | 'baseline_regression' | 'no_incremental_resolution' | 'incremental_regression' | 'firing_below_minimum' | 'firing_unobserved' | 'effect_below_minimum' | 'effect_unobserved' | 'cost_limit_exceeded';
|
|
2681
|
+
interface CrossSurfaceAdditionDecision {
|
|
2682
|
+
additionCandidateId: string;
|
|
2683
|
+
additionComponentId: string;
|
|
2684
|
+
bundleCandidateId: string | null;
|
|
2685
|
+
incrementalResolutionTaskIds: string[];
|
|
2686
|
+
incrementalRegressionTaskIds: string[];
|
|
2687
|
+
incrementalMedianCost: Record<string, number> | null;
|
|
2688
|
+
eligible: boolean;
|
|
2689
|
+
selected: boolean;
|
|
2690
|
+
reasons: CrossSurfaceAdditionRejectionReason[];
|
|
2691
|
+
}
|
|
2692
|
+
interface CrossSurfaceCompositionStep {
|
|
2693
|
+
fromCandidateId: string;
|
|
2694
|
+
retainedComponentIds: string[];
|
|
2695
|
+
considered: CrossSurfaceAdditionDecision[];
|
|
2696
|
+
selectedCandidateId: string | null;
|
|
2697
|
+
}
|
|
2698
|
+
/** One deterministic growth path starting from a compatible two-surface seed. */
|
|
2699
|
+
interface CrossSurfaceInteractionPath {
|
|
2700
|
+
seedCandidateId: string;
|
|
2701
|
+
terminalCandidateId: string;
|
|
2702
|
+
terminalComponentIds: string[];
|
|
2703
|
+
qualified: boolean;
|
|
2704
|
+
steps: CrossSurfaceCompositionStep[];
|
|
2705
|
+
}
|
|
2706
|
+
interface CrossSurfaceInteractionAwareSelection {
|
|
2707
|
+
/** Compatible pair that seeded the selected deterministic growth path. */
|
|
2708
|
+
seedCandidateId: string;
|
|
2709
|
+
/** Candidate reached by the winning path, even if the minimum size is not met. */
|
|
2710
|
+
terminalCandidateId: string;
|
|
2711
|
+
terminalComponentIds: string[];
|
|
2712
|
+
/** null when no path produced a qualifying multi-component bundle. */
|
|
2713
|
+
selectedCandidateId: string | null;
|
|
2714
|
+
qualified: boolean;
|
|
2715
|
+
/** Every compatible pair seed is retained so seed choice cannot hide an interaction. */
|
|
2716
|
+
evaluatedPaths: CrossSurfaceInteractionPath[];
|
|
2717
|
+
/** Convenience alias for the winning path's steps. */
|
|
2718
|
+
steps: CrossSurfaceCompositionStep[];
|
|
2719
|
+
}
|
|
2720
|
+
interface CrossSurfaceSelections {
|
|
2721
|
+
bestSingle: CrossSurfaceBestSingleSelection | null;
|
|
2722
|
+
naiveStack: CrossSurfaceNaiveStackSelection | null;
|
|
2723
|
+
interactionAware: CrossSurfaceInteractionAwareSelection | null;
|
|
2724
|
+
}
|
|
2725
|
+
interface CrossSurfaceInteractionReport<TRow extends CrossSurfaceTaskRow = CrossSurfaceTaskRow> {
|
|
2726
|
+
taskIds: string[];
|
|
2727
|
+
componentIds: string[];
|
|
2728
|
+
candidateIds: string[];
|
|
2729
|
+
costMetrics: string[];
|
|
2730
|
+
/** Canonical candidate × task order; no input row is dropped. */
|
|
2731
|
+
rows: TRow[];
|
|
2732
|
+
missingAttempts: TRow[];
|
|
2733
|
+
invalidAttempts: TRow[];
|
|
2734
|
+
candidates: CrossSurfaceCandidateSummary[];
|
|
2735
|
+
pairwise: CrossSurfacePairwiseEntry[];
|
|
2736
|
+
selections: CrossSurfaceSelections;
|
|
2737
|
+
}
|
|
2738
|
+
//#endregion
|
|
2739
|
+
//#region src/campaign/cross-surface-interaction.d.ts
|
|
2740
|
+
/**
|
|
2741
|
+
* Build the complete cross-surface evidence matrix and derive all three frozen
|
|
2742
|
+
* candidates. The task/candidate/component orders are part of the input so
|
|
2743
|
+
* neither insertion order nor an after-the-fact tie-break can change a result.
|
|
2744
|
+
*/
|
|
2745
|
+
declare function analyzeCrossSurfaceInteractions<TRow extends CrossSurfaceTaskRow>(input: AnalyzeCrossSurfaceInteractionsInput<TRow>): CrossSurfaceInteractionReport<TRow>;
|
|
2746
|
+
//#endregion
|
|
2747
|
+
//#region src/campaign/fixtures.d.ts
|
|
2748
|
+
type EvalFixtureValidationMode = 'vitest' | 'none';
|
|
2749
|
+
interface EvalFixtureFile {
|
|
2750
|
+
path: string;
|
|
2751
|
+
sha256: string;
|
|
2752
|
+
bytes: number;
|
|
2753
|
+
}
|
|
2754
|
+
interface EvalFixture {
|
|
2755
|
+
name: string;
|
|
2756
|
+
path: string;
|
|
2757
|
+
promptPath: string;
|
|
2758
|
+
evalPath?: string;
|
|
2759
|
+
packageJsonPath?: string;
|
|
2760
|
+
prompt: string;
|
|
2761
|
+
files: EvalFixtureFile[];
|
|
2762
|
+
fingerprint: string;
|
|
2763
|
+
}
|
|
2764
|
+
interface EvalFixtureScenario extends Scenario {
|
|
2765
|
+
kind: 'eval-fixture';
|
|
2766
|
+
fixtureName: string;
|
|
2767
|
+
fixturePath: string;
|
|
2768
|
+
promptPath: string;
|
|
2769
|
+
evalPath?: string;
|
|
2770
|
+
packageJsonPath?: string;
|
|
2771
|
+
prompt: string;
|
|
2772
|
+
fingerprint: string;
|
|
2773
|
+
}
|
|
2774
|
+
interface EvalFixtureLoadOptions {
|
|
2775
|
+
/** `vitest` requires EVAL.ts/EVAL.tsx and package.json type=module. `none` only requires PROMPT.md. */
|
|
2776
|
+
validation?: EvalFixtureValidationMode;
|
|
2777
|
+
/** Extra caller-owned knobs that affect fixture behavior, folded into the fingerprint. */
|
|
2778
|
+
fingerprintConfig?: unknown;
|
|
2779
|
+
}
|
|
2780
|
+
interface LoadEvalFixtureScenariosOptions extends EvalFixtureLoadOptions {
|
|
2781
|
+
names?: string[];
|
|
2782
|
+
}
|
|
2783
|
+
interface PlanEvalFixtureRunOptions<TArtifact = unknown> extends Pick<PlanCampaignRunOptions<EvalFixtureScenario, TArtifact>, 'dispatchRef' | 'judges' | 'seed' | 'reps' | 'resumable' | 'runDir'> {
|
|
2784
|
+
evalsDir: string;
|
|
2785
|
+
validation?: EvalFixtureValidationMode;
|
|
2786
|
+
fingerprintConfig?: unknown;
|
|
2787
|
+
names?: string[];
|
|
2788
|
+
storage?: CampaignStorage;
|
|
2789
|
+
}
|
|
2790
|
+
type EvalFixtureRunPlan = CampaignRunPlan & {
|
|
2791
|
+
fixtures: Array<Pick<EvalFixtureScenario, 'fixtureName' | 'fixturePath' | 'fingerprint'>>;
|
|
2792
|
+
};
|
|
2793
|
+
/** Walk `evalsDir` and return the relative name of every fixture directory (one containing an exact-case `PROMPT.md`). */
|
|
2794
|
+
declare function discoverEvalFixtures(evalsDir: string): string[];
|
|
2795
|
+
/**
|
|
2796
|
+
* Load ONE fixture by name: reads `PROMPT.md` (plus `EVAL.ts`/`EVAL.tsx` and `package.json` under
|
|
2797
|
+
* `vitest` validation) and content-fingerprints the full file set for cache identity.
|
|
2798
|
+
*/
|
|
2799
|
+
declare function loadEvalFixture(evalsDir: string, name: string, options?: EvalFixtureLoadOptions): EvalFixture;
|
|
2800
|
+
/** Load fixtures (all discovered, or just `names`) as campaign `Scenario`s tagged `eval-fixture`. */
|
|
2801
|
+
declare function loadEvalFixtureScenarios(evalsDir: string, options?: LoadEvalFixtureScenariosOptions): EvalFixtureScenario[];
|
|
2802
|
+
/**
|
|
2803
|
+
* Dry-run planner for a fixture campaign: loads the scenarios, delegates to `planCampaignRun`,
|
|
2804
|
+
* and returns the plan plus each fixture's name/path/fingerprint.
|
|
2805
|
+
*/
|
|
2806
|
+
declare function planEvalFixtureRun<TArtifact = unknown>(options: PlanEvalFixtureRunOptions<TArtifact>): EvalFixtureRunPlan;
|
|
2807
|
+
//#endregion
|
|
2808
|
+
//#region src/campaign/gates/neutralization-gate.d.ts
|
|
2809
|
+
interface NeutralizationGateOptions<TScenario extends Scenario = Scenario> {
|
|
2810
|
+
scenarios: TScenario[];
|
|
2811
|
+
/** Reject when the neutralized (content-blanked, footprint-matched) variant
|
|
2812
|
+
* reproduces at least this fraction of the candidate's held-out lift. Default
|
|
2813
|
+
* 0.5 — if blanking the content keeps half the lift, the content is decorative.
|
|
2814
|
+
* Equality rejects: a neutralized lift == threshold·candidateLift is decorative. */
|
|
2815
|
+
maxDecorativeFraction?: number;
|
|
2816
|
+
}
|
|
2817
|
+
/**
|
|
2818
|
+
* Composable placebo gate: ships only when the candidate's held-out lift is NOT
|
|
2819
|
+
* mostly reproduced by a footprint-matched neutralized variant.
|
|
2820
|
+
*/
|
|
2821
|
+
declare function neutralizationGate<TArtifact, TScenario extends Scenario>(options: NeutralizationGateOptions<TScenario>): Gate<TArtifact, TScenario>;
|
|
2822
|
+
//#endregion
|
|
2823
|
+
//#region src/campaign/grounded-reflection.d.ts
|
|
2824
|
+
/**
|
|
2825
|
+
* Evidence grounding for reflective optimizers (GEPA-style revise loops).
|
|
2826
|
+
*
|
|
2827
|
+
* Two failure modes recur when an LLM revises an artifact from raw rollout
|
|
2828
|
+
* traces (first measured in agent-lab R358, where naive reflection REGRESSED
|
|
2829
|
+
* the score 0.375 -> 0.125 before these helpers fixed it):
|
|
2830
|
+
*
|
|
2831
|
+
* 1. The environment often hides WHY a rollout failed - a tool call can
|
|
2832
|
+
* succeed while an invisible downstream check fails - so the reviser
|
|
2833
|
+
* cannot see the cause in the transcript. The only reliable signal is the
|
|
2834
|
+
* field-level difference between what passing and failing rollouts did.
|
|
2835
|
+
* `rolloutArgumentDiff` computes that difference deterministically so the
|
|
2836
|
+
* reviser is handed the diff instead of being trusted to derive it.
|
|
2837
|
+
*
|
|
2838
|
+
* 2. Revisers invent plausible-but-wrong literal values ("use 'new'",
|
|
2839
|
+
* "use 'sent'") that no passing rollout ever used, turning every rollout
|
|
2840
|
+
* into a failure. `classifyUngroundedLiterals` mechanically detects them,
|
|
2841
|
+
* separating HARMFUL literals (ones failing rollouts actually used -
|
|
2842
|
+
* proven damage) from benign illustrations (e.g. a name example like
|
|
2843
|
+
* 'Doe'), so callers can hard-reject the former and merely log the latter.
|
|
2844
|
+
* Rejecting every ungrounded quoted word is too blunt: it killed a run
|
|
2845
|
+
* over a surname illustration before the severity split existed.
|
|
2846
|
+
*
|
|
2847
|
+
* Pure data in, data out: no LLM calls, no filesystem, no domain knowledge.
|
|
2848
|
+
*/
|
|
2849
|
+
/** One tool/action call observed in a rollout: a name plus its arguments. */
|
|
2850
|
+
interface RolloutCall {
|
|
2851
|
+
readonly name: string;
|
|
2852
|
+
readonly args: Readonly<Record<string, unknown>>;
|
|
2853
|
+
}
|
|
2854
|
+
/** A scored rollout: its calls plus the scalar outcome used to split pass/fail. */
|
|
2855
|
+
interface ScoredRollout {
|
|
2856
|
+
/** Caller-meaningful identifier (task id, cell id) used only for reporting. */
|
|
2857
|
+
readonly id: string;
|
|
2858
|
+
/** Scalar outcome in [0, 1]; `passThreshold` splits passing from failing. */
|
|
2859
|
+
readonly score: number;
|
|
2860
|
+
readonly calls: readonly RolloutCall[];
|
|
2861
|
+
}
|
|
2862
|
+
interface RolloutArgumentDiffOptions {
|
|
2863
|
+
/** Rollouts with `score >= passThreshold` count as passing. Default 1. */
|
|
2864
|
+
readonly passThreshold?: number;
|
|
2865
|
+
/** Max distinct values listed per field per side in the rendered text. Default 4. */
|
|
2866
|
+
readonly maxValuesPerField?: number;
|
|
2867
|
+
}
|
|
2868
|
+
interface RolloutArgumentDiff {
|
|
2869
|
+
/** Human/LLM-readable per-field diff, one line per field. */
|
|
2870
|
+
readonly text: string;
|
|
2871
|
+
/** Lowercased stringified argument values seen in passing rollouts. */
|
|
2872
|
+
readonly passingValues: ReadonlySet<string>;
|
|
2873
|
+
/** Lowercased stringified argument values seen in failing rollouts. */
|
|
2874
|
+
readonly failingValues: ReadonlySet<string>;
|
|
2875
|
+
}
|
|
2876
|
+
/**
|
|
2877
|
+
* Deterministic per-field diff of call arguments between passing and failing
|
|
2878
|
+
* rollouts. A field set by failing rollouts but left unset by passing ones is
|
|
2879
|
+
* the classic poison-input signature; a field whose values differ across the
|
|
2880
|
+
* split points at the correct value. Feed `text` to the reviser verbatim.
|
|
2881
|
+
*/
|
|
2882
|
+
declare function rolloutArgumentDiff(rollouts: readonly ScoredRollout[], opts?: RolloutArgumentDiffOptions): RolloutArgumentDiff;
|
|
2883
|
+
interface UngroundedLiteralReport {
|
|
2884
|
+
/** Quoted single-word literals in the text that no passing rollout used. */
|
|
2885
|
+
readonly ungrounded: readonly string[];
|
|
2886
|
+
/** The subset failing rollouts actually used - prescribing these is proven harmful. */
|
|
2887
|
+
readonly harmful: readonly string[];
|
|
2888
|
+
}
|
|
2889
|
+
/**
|
|
2890
|
+
* Scan revised artifact text for single-quoted single-word literals (the
|
|
2891
|
+
* "use exactly 'new'" pattern) that appear in no passing rollout's argument
|
|
2892
|
+
* values. Multi-word quotes pass (they are prose, not prescriptions).
|
|
2893
|
+
* Callers should reject on `harmful` (with a bounded retry) and at most log
|
|
2894
|
+
* `ungrounded` - see the module header for why the severities differ.
|
|
2895
|
+
*/
|
|
2896
|
+
declare function classifyUngroundedLiterals(text: string, diff: Pick<RolloutArgumentDiff, 'passingValues' | 'failingValues'>): UngroundedLiteralReport;
|
|
2897
|
+
//#endregion
|
|
2898
|
+
//#region src/campaign/labeled-store/fs-adapter.d.ts
|
|
2899
|
+
interface FsLabeledScenarioStoreOptions {
|
|
2900
|
+
/** Root directory for JSONL files. Created if missing. */
|
|
2901
|
+
root: string;
|
|
2902
|
+
/** Per-source rate limit. When set, writes exceeding the cap are rejected
|
|
2903
|
+
* with a typed error. Default: no limit. */
|
|
2904
|
+
maxWritesPerMinutePerBucket?: number;
|
|
2905
|
+
/** Test seam — override `Date.now()` for deterministic tests. */
|
|
2906
|
+
now?: () => number;
|
|
2907
|
+
}
|
|
2908
|
+
/** Typed rejection from a labeled-scenario store (bad provenance, rate limit, invalid sample args) — carries a stable string `code`. */
|
|
2909
|
+
declare class LabeledScenarioStoreError extends Error {
|
|
2910
|
+
readonly code: string;
|
|
2911
|
+
constructor(code: string, message: string);
|
|
2912
|
+
}
|
|
2913
|
+
/**
|
|
2914
|
+
* Filesystem `LabeledScenarioStore`: appends one JSONL file per source with provenance and
|
|
2915
|
+
* rate-limit guards. For tests, local dev, and small workloads — high-throughput lands in Turso.
|
|
2916
|
+
*/
|
|
2917
|
+
declare class FsLabeledScenarioStore implements LabeledScenarioStore {
|
|
2918
|
+
private readonly options;
|
|
2919
|
+
private readonly now;
|
|
2920
|
+
private readonly rateLimits;
|
|
2921
|
+
constructor(options: FsLabeledScenarioStoreOptions);
|
|
2922
|
+
observe(write: LabeledScenarioWrite): Promise<void>;
|
|
2923
|
+
sample(args: LabeledScenarioSampleArgs): Promise<LabeledScenarioRecord[]>;
|
|
2924
|
+
size(): Promise<{
|
|
2925
|
+
train: number;
|
|
2926
|
+
test: number;
|
|
2927
|
+
bySource: Record<string, number>;
|
|
2928
|
+
byTrust: Record<LabelTrust, number>;
|
|
2929
|
+
}>;
|
|
2930
|
+
private assertProvenance;
|
|
2931
|
+
private assertRateLimit;
|
|
2932
|
+
private toRecord;
|
|
2933
|
+
private pathForSource;
|
|
2934
|
+
}
|
|
2935
|
+
//#endregion
|
|
2936
|
+
//#region src/campaign/neutralize.d.ts
|
|
2937
|
+
/**
|
|
2938
|
+
* @module
|
|
2939
|
+
* Footprint-matched neutralization — the placebo control for content-vs-footprint
|
|
2940
|
+
* attribution in a promotion gate.
|
|
2941
|
+
*
|
|
2942
|
+
* A promoted surface can raise a held-out score two different ways:
|
|
2943
|
+
* 1. its CONTENT is informative (the thing we want to promote), or
|
|
2944
|
+
* 2. it merely added prompt/mount FOOTPRINT — more bytes, more lines, a longer
|
|
2945
|
+
* more authoritative-looking prompt — that the model spends attention on
|
|
2946
|
+
* regardless of what the bytes say.
|
|
2947
|
+
*
|
|
2948
|
+
* A held-out gate proves the candidate beat baseline; it cannot separate (1) from
|
|
2949
|
+
* (2). `neutralizeText` produces a variant that keeps the input's layout and
|
|
2950
|
+
* length while carrying ZERO information, so scoring it isolates the footprint
|
|
2951
|
+
* contribution (2). Feed the neutralized variant's scores to `neutralizationGate`:
|
|
2952
|
+
* any lift it still holds over baseline is decorative, and a candidate whose lift
|
|
2953
|
+
* survives neutralization is rejected however large its raw lift.
|
|
2954
|
+
*/
|
|
2955
|
+
/**
|
|
2956
|
+
* Blank every non-whitespace character to a 1-byte filler while preserving all
|
|
2957
|
+
* whitespace. Line count, indentation, and word/line lengths are unchanged — so
|
|
2958
|
+
* the neutralized variant has the same layout and (for ASCII) the same byte
|
|
2959
|
+
* footprint as the input, but no readable content. Whitespace is preserved
|
|
2960
|
+
* deliberately: collapsing it would change the token structure and stop the
|
|
2961
|
+
* variant from being a true footprint match.
|
|
2962
|
+
*/
|
|
2963
|
+
declare function neutralizeText(content: string): string;
|
|
2964
|
+
//#endregion
|
|
2965
|
+
//#region src/artifact-validator.d.ts
|
|
2966
|
+
/**
|
|
2967
|
+
* Artifact validators.
|
|
2968
|
+
*
|
|
2969
|
+
* Generic "score a produced artifact" primitive. Tax uses it for PDF form
|
|
2970
|
+
* correctness, research for sourced briefs, browser for task assertions, coding
|
|
2971
|
+
* for social posts. One interface, many validators.
|
|
2972
|
+
*
|
|
2973
|
+
* A validator receives an `Artifact` (file on disk, JSON blob, text, binary)
|
|
2974
|
+
* plus a `ValidationContext` (scenario id, the turns that produced it) and
|
|
2975
|
+
* returns a `ValidationResult` with pass/fail + 0..1 score + structured
|
|
2976
|
+
* issues.
|
|
2977
|
+
*/
|
|
2978
|
+
interface Artifact {
|
|
2979
|
+
/** Logical kind — validators type-guard on this */
|
|
2980
|
+
kind: 'file' | 'json' | 'text' | 'binary' | string;
|
|
2981
|
+
/** Filesystem-style path, optional */
|
|
2982
|
+
path?: string;
|
|
2983
|
+
/** String content for text/json/file kinds */
|
|
2984
|
+
content?: string;
|
|
2985
|
+
/** Binary content (if kind === 'binary') */
|
|
2986
|
+
bytes?: Uint8Array;
|
|
2987
|
+
/** Caller-supplied metadata (mimeType, sha256, size, etc.) */
|
|
2988
|
+
metadata?: Record<string, unknown>;
|
|
2989
|
+
}
|
|
2990
|
+
//#endregion
|
|
2991
|
+
//#region src/completion-verifier.d.ts
|
|
2992
|
+
/** What kind of produced state can satisfy a requirement structurally. */
|
|
2993
|
+
type SatisfiedBy = 'artifact' | 'proposal' | 'tool-call' | 'any';
|
|
2994
|
+
interface CompletionRequirement {
|
|
2995
|
+
/** Stable id from the task gold (e.g. a persona's `expected_requirements[].req_id`). */
|
|
2996
|
+
reqId: string;
|
|
2997
|
+
/** Human-readable description of the required deliverable. */
|
|
2998
|
+
title: string;
|
|
2999
|
+
/** Optional kind/category hint, matched against a produced item's kind. */
|
|
3000
|
+
category?: string;
|
|
3001
|
+
/** What produced state satisfies this requirement. Defaults to 'any'. */
|
|
3002
|
+
satisfiedBy?: SatisfiedBy;
|
|
3003
|
+
}
|
|
3004
|
+
interface TaskGold {
|
|
3005
|
+
taskId: string;
|
|
3006
|
+
requirements: CompletionRequirement[];
|
|
3007
|
+
}
|
|
3008
|
+
interface ProducedProposal {
|
|
3009
|
+
id: string;
|
|
3010
|
+
title: string;
|
|
3011
|
+
status: 'pending' | 'approved' | 'rejected';
|
|
3012
|
+
/** Optional persisted body — when present, enables a correctness check. */
|
|
3013
|
+
content?: string;
|
|
3014
|
+
}
|
|
3015
|
+
/** Everything observable about what a run actually produced. */
|
|
3016
|
+
interface ProducedState {
|
|
3017
|
+
/** Persisted vault artifacts. Reuses the shared `Artifact` shape. */
|
|
3018
|
+
artifacts: Artifact[];
|
|
3019
|
+
/** Proposals / filings the agent created. */
|
|
3020
|
+
proposals: ProducedProposal[];
|
|
3021
|
+
/** Names of tools the agent invoked. */
|
|
3022
|
+
toolCalls: string[];
|
|
3023
|
+
}
|
|
3024
|
+
interface RequirementCheck {
|
|
3025
|
+
reqId: string;
|
|
3026
|
+
title: string;
|
|
3027
|
+
/** A produced item of the right kind matched the requirement, non-empty. */
|
|
3028
|
+
structurallyPresent: boolean;
|
|
3029
|
+
/**
|
|
3030
|
+
* Whether the matched item actually fulfils the requirement. `null` when
|
|
3031
|
+
* not structurally present, when the matched item carries no content
|
|
3032
|
+
* to assess, or when the correctness check itself failed (`unmeasured`).
|
|
3033
|
+
*/
|
|
3034
|
+
correct: boolean | null;
|
|
3035
|
+
/** structurallyPresent && !unmeasured && correct !== false. */
|
|
3036
|
+
satisfied: boolean;
|
|
3037
|
+
/**
|
|
3038
|
+
* Set when the correctness check itself errored (LLM call failure or an
|
|
3039
|
+
* unparseable response after retry). The requirement's fulfilment is
|
|
3040
|
+
* UNKNOWN — `correct` stays null, `satisfied` is false, and
|
|
3041
|
+
* `completionVerdict` excludes the row from `completionRate`'s
|
|
3042
|
+
* denominator. Never folded into a zero: a synthetic zero is
|
|
3043
|
+
* indistinguishable from a real failure (see `JudgeParseError`).
|
|
3044
|
+
*/
|
|
3045
|
+
unmeasured?: true;
|
|
3046
|
+
/** Why the correctness check could not be measured (present iff `unmeasured`). */
|
|
3047
|
+
unmeasuredReason?: string;
|
|
3048
|
+
/** Human-readable evidence for the verdict. */
|
|
3049
|
+
evidence: string[];
|
|
3050
|
+
}
|
|
3051
|
+
/** Extends the substrate verdict spine: `valid` = `fullyComplete` and
|
|
3052
|
+
* `score` = `completionRate` — derived in `completionVerdict()`, the one
|
|
3053
|
+
* place those equalities hold by construction. */
|
|
3054
|
+
interface CompletionVerdict extends DefaultVerdict {
|
|
3055
|
+
taskId: string;
|
|
3056
|
+
requirements: RequirementCheck[];
|
|
3057
|
+
/** satisfied / MEASURABLE requirements (unmeasured rows leave the denominator). */
|
|
3058
|
+
completionRate: number;
|
|
3059
|
+
/** Every measurable requirement satisfied (false when anything is unmeasured). */
|
|
3060
|
+
fullyComplete: boolean;
|
|
3061
|
+
/** Requirements whose correctness check errored — reported, never scored as zero. */
|
|
3062
|
+
unmeasuredCount: number;
|
|
3063
|
+
}
|
|
3064
|
+
/**
|
|
3065
|
+
* Construct a `CompletionVerdict` from the per-requirement checks, deriving
|
|
3066
|
+
* `completionRate` / `fullyComplete` and the spine fields (`valid` =
|
|
3067
|
+
* `fullyComplete`, `score` = `completionRate`) in one place. Throws on zero
|
|
3068
|
+
* requirements — a verdict over nothing is a misconfiguration, mirroring
|
|
3069
|
+
* `verifyCompletion`'s gold-spec guard.
|
|
3070
|
+
*/
|
|
3071
|
+
declare function completionVerdict(input: {
|
|
3072
|
+
taskId: string;
|
|
3073
|
+
requirements: RequirementCheck[];
|
|
3074
|
+
/** What certified the correctness stage, when anything did. Omitted =
|
|
3075
|
+
* an uncertified verdict — the honest default for a bare checker. */
|
|
3076
|
+
certification?: VerdictCertification;
|
|
3077
|
+
}): CompletionVerdict;
|
|
3078
|
+
/**
|
|
3079
|
+
* What a correctness checker declares about itself so the completion
|
|
3080
|
+
* verdict can carry a certification: which strategy member it discharges,
|
|
3081
|
+
* its exact identity, and the steps its answers rest on unverified.
|
|
3082
|
+
*/
|
|
3083
|
+
interface CorrectnessCheckerAttestation {
|
|
3084
|
+
strategy: VerificationStrategySource;
|
|
3085
|
+
checker: CheckerIdentity;
|
|
3086
|
+
assumptions: string[];
|
|
3087
|
+
}
|
|
3088
|
+
/**
|
|
3089
|
+
* Decides whether a produced item's content actually fulfils a requirement.
|
|
3090
|
+
* Injected so the structural verifier stays pure and unit-testable; the
|
|
3091
|
+
* production implementation is `createLlmCorrectnessChecker`.
|
|
3092
|
+
*
|
|
3093
|
+
* `attestation` is optional metadata on the function value: a checker that
|
|
3094
|
+
* carries one yields certified completion verdicts; a bare function yields
|
|
3095
|
+
* the same verdict uncertified. A plain arrow function remains a valid
|
|
3096
|
+
* checker.
|
|
3097
|
+
*/
|
|
3098
|
+
interface CorrectnessChecker {
|
|
3099
|
+
(requirement: CompletionRequirement, content: string): Promise<{
|
|
3100
|
+
correct: boolean;
|
|
3101
|
+
reason: string;
|
|
3102
|
+
}>;
|
|
3103
|
+
attestation?: CorrectnessCheckerAttestation;
|
|
3104
|
+
}
|
|
3105
|
+
/**
|
|
3106
|
+
* Verify whether a run completed the task. `checkCorrectness` is injected —
|
|
3107
|
+
* `createLlmCorrectnessChecker` for production, a deterministic stub in tests.
|
|
3108
|
+
*
|
|
3109
|
+
* Throws on a gold spec with no requirements: an eval task that requires
|
|
3110
|
+
* nothing is a misconfiguration, not a vacuously-complete task.
|
|
3111
|
+
*/
|
|
3112
|
+
declare function verifyCompletion(gold: TaskGold, state: ProducedState, checkCorrectness: CorrectnessChecker): Promise<CompletionVerdict>;
|
|
3113
|
+
interface LlmCorrectnessCheckerOpts {
|
|
3114
|
+
model?: string;
|
|
3115
|
+
/** Optional ledger for direct use. */
|
|
3116
|
+
costLedger?: CostLedgerHandle;
|
|
3117
|
+
costPhase?: string;
|
|
3118
|
+
costTags?: Record<string, string>;
|
|
3119
|
+
signal?: AbortSignal;
|
|
3120
|
+
/** Max chars of artifact content sent to the checker. */
|
|
3121
|
+
maxContentChars?: number;
|
|
3122
|
+
/**
|
|
3123
|
+
* Checker LLM calls per requirement before giving up (parse failures and
|
|
3124
|
+
* call errors both consume attempts). The failure then surfaces as an
|
|
3125
|
+
* `unmeasured` requirement, never a zero.
|
|
3126
|
+
*/
|
|
3127
|
+
maxAttempts?: number;
|
|
3128
|
+
/**
|
|
3129
|
+
* Forensic capture of every checker request/response/error — without it a
|
|
3130
|
+
* checker failure is unauditable (the agent-turn raws never contain the
|
|
3131
|
+
* checker's own calls). Same sink contract as `LlmClient`.
|
|
3132
|
+
*/
|
|
3133
|
+
rawSink?: RawProviderSink;
|
|
3134
|
+
}
|
|
3135
|
+
/**
|
|
3136
|
+
* Production `CorrectnessChecker` — one LLM call per matched artifact,
|
|
3137
|
+
* deterministic (temperature 0), structured JSON out. Judges fulfilment
|
|
3138
|
+
* only: a plan, a gesture, or a description of what should be done does not
|
|
3139
|
+
* fulfil a requirement — the artifact must BE the deliverable.
|
|
3140
|
+
*/
|
|
3141
|
+
declare function createLlmCorrectnessChecker(chat: ChatClient, opts?: LlmCorrectnessCheckerOpts): CorrectnessChecker;
|
|
3142
|
+
//#endregion
|
|
3143
|
+
//#region src/produced-state.d.ts
|
|
3144
|
+
/** A tool the agent invoked. */
|
|
3145
|
+
interface ToolCallEventLike {
|
|
3146
|
+
type: 'tool_call';
|
|
3147
|
+
toolName: string;
|
|
3148
|
+
}
|
|
3149
|
+
/**
|
|
3150
|
+
* An artifact the agent produced. `content` is the enriched field — the
|
|
3151
|
+
* runtime's base `artifact` event carries only metadata; the completion
|
|
3152
|
+
* oracle needs the body to verify the deliverable, so the runtime emits it.
|
|
3153
|
+
*/
|
|
3154
|
+
interface ArtifactEventLike {
|
|
3155
|
+
type: 'artifact';
|
|
3156
|
+
artifactId: string;
|
|
3157
|
+
name?: string;
|
|
3158
|
+
mimeType?: string;
|
|
3159
|
+
uri?: string;
|
|
3160
|
+
content?: string;
|
|
3161
|
+
}
|
|
3162
|
+
/** A proposal / filing the agent created. */
|
|
3163
|
+
interface ProposalEventLike {
|
|
3164
|
+
type: 'proposal_created';
|
|
3165
|
+
proposalId: string;
|
|
3166
|
+
title: string;
|
|
3167
|
+
status?: 'pending' | 'approved' | 'rejected';
|
|
3168
|
+
content?: string;
|
|
3169
|
+
}
|
|
3170
|
+
/**
|
|
3171
|
+
* The subset of runtime stream events `extractProducedState` consumes.
|
|
3172
|
+
* agent-runtime's full `RuntimeStreamEvent` union satisfies this structurally;
|
|
3173
|
+
* the `{ type: string }` catch-all keeps the input permissive so callers can
|
|
3174
|
+
* pass the whole unfiltered telemetry stream — unrecognized events are skipped.
|
|
3175
|
+
*/
|
|
3176
|
+
type RuntimeEventLike = ToolCallEventLike | ArtifactEventLike | ProposalEventLike | {
|
|
3177
|
+
type: string;
|
|
3178
|
+
};
|
|
3179
|
+
/**
|
|
3180
|
+
* Normalize a run's runtime event stream into `ProducedState`.
|
|
3181
|
+
*
|
|
3182
|
+
* Pure and total — unrecognized event types are skipped. `toolCalls` is
|
|
3183
|
+
* deduplicated by name in first-seen order (completion cares about a tool's
|
|
3184
|
+
* presence, not its call count). An artifact with neither a name nor a uri
|
|
3185
|
+
* still yields an entry keyed by its `artifactId` so it is never silently
|
|
3186
|
+
* dropped; an artifact with no `content` yields empty content, which the
|
|
3187
|
+
* completion oracle's structural check then rejects on its own.
|
|
3188
|
+
*/
|
|
3189
|
+
declare function extractProducedState(events: readonly RuntimeEventLike[]): ProducedState;
|
|
3190
|
+
//#endregion
|
|
3191
|
+
//#region src/integrity/backend-integrity.d.ts
|
|
3192
|
+
interface BackendIntegrityReport {
|
|
3193
|
+
/** Total records inspected. */
|
|
3194
|
+
totalRecords: number;
|
|
3195
|
+
/** Records with input=0 AND output=0 (a stub fingerprint). */
|
|
3196
|
+
stubRecords: number;
|
|
3197
|
+
/** Records with nonzero token usage (real LLM activity). */
|
|
3198
|
+
realRecords: number;
|
|
3199
|
+
/** Records where output>0 but costUsd=0 (real LLM, broken cost ledger). */
|
|
3200
|
+
uncostedRecords: number;
|
|
3201
|
+
/** Sum of input tokens across all records. */
|
|
3202
|
+
totalInputTokens: number;
|
|
3203
|
+
/** Sum of output tokens across all records. */
|
|
3204
|
+
totalOutputTokens: number;
|
|
3205
|
+
/** Sum of costUsd across all records. */
|
|
3206
|
+
totalCostUsd: number;
|
|
3207
|
+
/** Worst-case integrity verdict. */
|
|
3208
|
+
verdict: 'real' | 'mixed' | 'stub';
|
|
3209
|
+
/** Human-readable diagnosis suitable for terminal output. */
|
|
3210
|
+
diagnosis: string;
|
|
3211
|
+
}
|
|
3212
|
+
/**
|
|
3213
|
+
* Error thrown when an integrity assertion fails. Caller can pattern-match
|
|
3214
|
+
* by `code === 'AGENT_EVAL_BACKEND_STUB'` to differentiate from other
|
|
3215
|
+
* errors.
|
|
3216
|
+
*/
|
|
3217
|
+
declare class BackendIntegrityError extends AgentEvalError {
|
|
3218
|
+
readonly report: BackendIntegrityReport;
|
|
3219
|
+
constructor(message: string, report: BackendIntegrityReport);
|
|
3220
|
+
}
|
|
3221
|
+
/**
|
|
3222
|
+
* Inspect a batch of RunRecords and return an integrity report. Pure
|
|
3223
|
+
* function — no I/O, no logging. The caller decides what to do with the
|
|
3224
|
+
* verdict (print warning, throw, gate CI, etc.).
|
|
3225
|
+
*/
|
|
3226
|
+
declare function summarizeBackendIntegrity(records: ReadonlyArray<RunRecord>): BackendIntegrityReport;
|
|
3227
|
+
/**
|
|
3228
|
+
* Throw BackendIntegrityError if the verdict is 'stub' — i.e. every record
|
|
3229
|
+
* shows zero LLM activity. Non-strict callers can pass `{ allowMixed: false }`
|
|
3230
|
+
* to also reject mixed verdicts (recommended for CI gates).
|
|
3231
|
+
*
|
|
3232
|
+
* Real backends pass through silently.
|
|
3233
|
+
*/
|
|
3234
|
+
declare function assertRealBackend(records: ReadonlyArray<RunRecord>, opts?: {
|
|
3235
|
+
allowMixed?: boolean;
|
|
3236
|
+
}): BackendIntegrityReport;
|
|
3237
|
+
//#endregion
|
|
3238
|
+
//#region src/campaign/presets/run-profile-matrix.d.ts
|
|
3239
|
+
/** Thrown when the matrix is misconfigured (no profiles, missing resolved model evidence,
|
|
3240
|
+
* etc.). Distinct from `BackendIntegrityError`,
|
|
3241
|
+
* which signals a stub backend at run time. */
|
|
3242
|
+
declare class ProfileMatrixError extends AgentEvalError {
|
|
3243
|
+
constructor(message: string);
|
|
3244
|
+
}
|
|
3245
|
+
/** Dispatch for one cell: render `profile` against `scenario`, returning the
|
|
3246
|
+
* artifact the judges score. Run LLM work through `ctx.cost.runPaidCall` —
|
|
3247
|
+
* the integrity check depends on its receipt. */
|
|
3248
|
+
type ProfileDispatchFn<TScenario extends Scenario, TArtifact> = (profile: AgentProfile$1, scenario: TScenario, ctx: DispatchContext) => Promise<TArtifact>;
|
|
3249
|
+
interface RunProfileMatrixOptions<TScenario extends Scenario, TArtifact> {
|
|
3250
|
+
/** Axis 3 — the agent-under-test configurations. Each is one column. */
|
|
3251
|
+
profiles: AgentProfile$1[];
|
|
3252
|
+
/** Axis 1 — the persona/scenario corpus, run against every profile. */
|
|
3253
|
+
scenarios: TScenario[];
|
|
3254
|
+
/** Renders one (profile, scenario) cell. */
|
|
3255
|
+
dispatch: ProfileDispatchFn<TScenario, TArtifact>;
|
|
3256
|
+
/** The scoring axis. */
|
|
3257
|
+
judges?: JudgeConfig<TArtifact, TScenario>[];
|
|
3258
|
+
/** Where each profile's campaign writes artifacts/traces. One subdir per
|
|
3259
|
+
* profile. */
|
|
3260
|
+
runDir: string;
|
|
3261
|
+
/** Git SHA the harness ran from — stamped onto every RunRecord (mandatory
|
|
3262
|
+
* for paper-grade records). */
|
|
3263
|
+
commitSha: string;
|
|
3264
|
+
/** Additional stable identity for dispatch behavior that can change without
|
|
3265
|
+
* changing `commitSha`, such as a caller-owned executable or remote config. */
|
|
3266
|
+
dispatchRef?: string;
|
|
3267
|
+
/** Logical experiment id shared across the whole matrix so the promotion
|
|
3268
|
+
* gate can pair profiles on matched scenarios. Default: a hash of the
|
|
3269
|
+
* profile + scenario ids. */
|
|
3270
|
+
experimentId?: string;
|
|
3271
|
+
/** Which split these runs belong to. Default `'search'`. */
|
|
3272
|
+
splitTag?: RunSplitTag;
|
|
3273
|
+
/** Replicates per (profile, scenario) cell for CI bands. Default 1. */
|
|
3274
|
+
reps?: number;
|
|
3275
|
+
/** Campaign seed (per profile). Default 42. */
|
|
3276
|
+
seed?: number;
|
|
3277
|
+
/**
|
|
3278
|
+
* Backend-integrity posture, enforced AFTER the matrix completes:
|
|
3279
|
+
* - `'assert'` (default) — throw `BackendIntegrityError` if the run was a
|
|
3280
|
+
* stub (and, with `allowMixed:false`, if it was mixed).
|
|
3281
|
+
* - `'warn'` — log the verdict but never throw.
|
|
3282
|
+
* - `'off'` — skip the guard entirely (only for offline/replay analysis).
|
|
3283
|
+
*/
|
|
3284
|
+
integrity?: 'assert' | 'warn' | 'off';
|
|
3285
|
+
/** Forwarded to `assertRealBackend`. Default true (tolerate partial 429
|
|
3286
|
+
* cascades); set false for strict CI gates. */
|
|
3287
|
+
allowMixed?: boolean;
|
|
3288
|
+
/** Max concurrent cells WITHIN each profile's campaign. Default 2. */
|
|
3289
|
+
maxConcurrency?: number;
|
|
3290
|
+
/** Max profile campaigns in flight. Default 1. Each profile keeps its own
|
|
3291
|
+
* run directory and cost ceiling; raise this when those resources are independent. */
|
|
3292
|
+
maxProfileConcurrency?: number;
|
|
3293
|
+
/** Cumulative USD cap per profile campaign. */
|
|
3294
|
+
costCeiling?: number;
|
|
3295
|
+
/** Capture flywheel — forwarded to each campaign. */
|
|
3296
|
+
labeledStore?: LabeledScenarioStore | 'off';
|
|
3297
|
+
captureSource?: LabeledScenarioSource;
|
|
3298
|
+
/** Storage backend. Default `fsCampaignStorage`. Pass
|
|
3299
|
+
* `inMemoryCampaignStorage()` for edge/CF-Worker/test runs. */
|
|
3300
|
+
storage?: CampaignStorage;
|
|
3301
|
+
/** Test seam — override the wall clock. */
|
|
3302
|
+
now?: () => Date;
|
|
3303
|
+
/** Optional persona key per scenario — drives the `byPersona` pivot. When
|
|
3304
|
+
* unset, `byPersona` is omitted. */
|
|
3305
|
+
personaOf?: (scenario: TScenario) => string;
|
|
3306
|
+
/** Validate every produced RunRecord with `validateRunRecord` (fail-loud).
|
|
3307
|
+
* Default true — catches bad model snapshots and non-finite judge dims at
|
|
3308
|
+
* the boundary instead of letting them poison downstream analysis. */
|
|
3309
|
+
validate?: boolean;
|
|
3310
|
+
/** Corpus-by-default: derive the trajectory text (`prompt` + `completion`)
|
|
3311
|
+
* for each cell from its artifact + scenario. When set, every produced
|
|
3312
|
+
* record carries `prompt`/`completion` (a `CorpusRecord`) so the run's
|
|
3313
|
+
* graded trajectories can be appended to the durable RL corpus with no
|
|
3314
|
+
* side-channel — `appendToCorpus(result.records, path)`. Fail-soft: a
|
|
3315
|
+
* throwing or undefined-returning extractor just omits the text. */
|
|
3316
|
+
corpusText?: (artifact: TArtifact, scenario: TScenario) => {
|
|
3317
|
+
prompt: string;
|
|
3318
|
+
completion: string;
|
|
3319
|
+
} | undefined;
|
|
3320
|
+
/**
|
|
3321
|
+
* Optional explicit row selection. The matrix identity remains based on the
|
|
3322
|
+
* complete profiles × scenarios × reps design; this invocation executes only
|
|
3323
|
+
* rows accepted by the predicate.
|
|
3324
|
+
*/
|
|
3325
|
+
rowFilter?: (input: {
|
|
3326
|
+
profile: AgentProfile$1;
|
|
3327
|
+
scenario: TScenario;
|
|
3328
|
+
rep: number;
|
|
3329
|
+
}) => boolean;
|
|
3330
|
+
/** Stable matrix identity supplied by a persisted profile-matrix plan. */
|
|
3331
|
+
matrixId?: string;
|
|
3332
|
+
/** Reuse cached failed cells. Normal matrix runs retry them by default. */
|
|
3333
|
+
reuseFailedCells?: boolean;
|
|
3334
|
+
}
|
|
3335
|
+
interface ProfileSummary {
|
|
3336
|
+
profileId: string;
|
|
3337
|
+
profileHash: string;
|
|
3338
|
+
model: string;
|
|
3339
|
+
/** RunRecords produced for this profile (= scenarios × reps). */
|
|
3340
|
+
records: number;
|
|
3341
|
+
/** Mean across scored records, or null when the profile has no task labels. */
|
|
3342
|
+
meanComposite: number | null;
|
|
3343
|
+
/** Total cost, or null when any call's cost was not captured. */
|
|
3344
|
+
totalCostUsd: number | null;
|
|
3345
|
+
costProvenance: CostProvenance;
|
|
3346
|
+
/** Per-profile integrity verdict — surfaces a single profile that ran stub
|
|
3347
|
+
* even when the matrix as a whole looks real. */
|
|
3348
|
+
integrity: BackendIntegrityReport;
|
|
3349
|
+
}
|
|
3350
|
+
interface ScenarioRollup {
|
|
3351
|
+
meanComposite: number;
|
|
3352
|
+
n: number;
|
|
3353
|
+
}
|
|
3354
|
+
interface RunProfileMatrixResult<TArtifact, TScenario extends Scenario> {
|
|
3355
|
+
matrixId: string;
|
|
3356
|
+
experimentId: string;
|
|
3357
|
+
/** One RunRecord per (profile, scenario, rep) cell — the integrity-checked,
|
|
3358
|
+
* paper-grade output. Feed straight into `analyzeRuns`, `HeldOutGate`,
|
|
3359
|
+
* scorecards, the hosted wire format. */
|
|
3360
|
+
records: RunRecord[];
|
|
3361
|
+
byProfile: Record<string, ProfileSummary>;
|
|
3362
|
+
byScenario: Record<string, ScenarioRollup>;
|
|
3363
|
+
/** Present only when `personaOf` was supplied. */
|
|
3364
|
+
byPersona?: Record<string, ScenarioRollup>;
|
|
3365
|
+
/** Whole-matrix integrity report (the one `integrity:'assert'` enforces). */
|
|
3366
|
+
integrity: BackendIntegrityReport;
|
|
3367
|
+
/** The raw per-profile campaign results, keyed by profile id. */
|
|
3368
|
+
campaigns: Record<string, CampaignResult<TArtifact, TScenario>>;
|
|
3369
|
+
}
|
|
3370
|
+
/**
|
|
3371
|
+
* Profile × scenario matrix runner: fan N agent profiles across M scenarios, project each cell to a validated `RunRecord` with real token usage, and enforce the backend-integrity guard before returning.
|
|
3372
|
+
*/
|
|
3373
|
+
declare function runProfileMatrix<TScenario extends Scenario, TArtifact>(opts: RunProfileMatrixOptions<TScenario, TArtifact>): Promise<RunProfileMatrixResult<TArtifact, TScenario>>;
|
|
3374
|
+
//#endregion
|
|
3375
|
+
//#region src/campaign/presets/playback.d.ts
|
|
3376
|
+
/** One step of a user story — what the user does. The driver interprets
|
|
3377
|
+
* `payload` (a Playwright selector + action, or a sandbox chat turn). */
|
|
3378
|
+
interface PlaybackStep {
|
|
3379
|
+
/** Human-readable action, captured verbatim in the UX narrative. */
|
|
3380
|
+
action: string;
|
|
3381
|
+
/** Driver-specific payload (e.g. `{ selector, fill }` or `{ turn }`). */
|
|
3382
|
+
payload?: Record<string, unknown>;
|
|
3383
|
+
}
|
|
3384
|
+
/**
|
|
3385
|
+
* A user story = a runnable product journey plus the requirements that define
|
|
3386
|
+
* "this story works". Each requirement is one Jira ticket line. Extends
|
|
3387
|
+
* `Scenario` so a catalog drops straight into `runProfileMatrix({ scenarios })`.
|
|
3388
|
+
*/
|
|
3389
|
+
interface UserStory extends Scenario {
|
|
3390
|
+
/** Human-readable story title (the ticket headline). */
|
|
3391
|
+
title: string;
|
|
3392
|
+
/** Ordered steps the driver executes. */
|
|
3393
|
+
steps: PlaybackStep[];
|
|
3394
|
+
/** What must hold in the produced state for the story to pass. */
|
|
3395
|
+
requirements: CompletionRequirement[];
|
|
3396
|
+
}
|
|
3397
|
+
/** Dispatch context plus the profile under test (which cheap model, etc.). */
|
|
3398
|
+
interface PlaybackContext extends DispatchContext {
|
|
3399
|
+
profile: AgentProfile;
|
|
3400
|
+
}
|
|
3401
|
+
/**
|
|
3402
|
+
* Drives the real product through a story and returns the runtime event stream
|
|
3403
|
+
* `extractProducedState` consumes. Implemented by CONSUMERS —
|
|
3404
|
+
* `SandboxPlaybackDriver` (real API / sandbox workspace) and
|
|
3405
|
+
* `PlaywrightPlaybackDriver` (real UI) — because they depend on runtime /
|
|
3406
|
+
* browser infra the substrate must not import. The driver MUST report LLM
|
|
3407
|
+
* usage through `ctx.cost.runPaidCall` so the backend-integrity check sees real
|
|
3408
|
+
* tokens (a run that never reports tokens reads as a stub).
|
|
3409
|
+
*/
|
|
3410
|
+
interface PlaybackDriver<TStory extends UserStory = UserStory> {
|
|
3411
|
+
run(story: TStory, ctx: PlaybackContext): Promise<readonly RuntimeEventLike[]>;
|
|
3412
|
+
}
|
|
3413
|
+
/**
|
|
3414
|
+
* Adapt a `PlaybackDriver` into a `runProfileMatrix` dispatch. The artifact the
|
|
3415
|
+
* matrix scores is the `ProducedState` extracted from the driver's event
|
|
3416
|
+
* stream — grade it with `scoreUserStory` (or a judge wrapping it).
|
|
3417
|
+
*/
|
|
3418
|
+
declare function makePlaybackDispatch<TStory extends UserStory>(driver: PlaybackDriver<TStory>): ProfileDispatchFn<TStory, ProducedState>;
|
|
3419
|
+
/** A scored user story — the completion verdict plus its human title. */
|
|
3420
|
+
interface UserStoryVerdict extends CompletionVerdict {
|
|
3421
|
+
title: string;
|
|
3422
|
+
}
|
|
3423
|
+
/**
|
|
3424
|
+
* Score one story's produced state against its requirements. Thin wrapper over
|
|
3425
|
+
* `verifyCompletion` that builds the gold from the story and returns a
|
|
3426
|
+
* per-requirement PASS/FAIL verdict. `checkCorrectness` is injected — a
|
|
3427
|
+
* deterministic stub in tests, `createLlmCorrectnessChecker` in production.
|
|
3428
|
+
*/
|
|
3429
|
+
declare function scoreUserStory(story: UserStory, state: ProducedState, checkCorrectness: CorrectnessChecker): Promise<UserStoryVerdict>;
|
|
3430
|
+
/** One row of the launch scoreboard — story × requirement → PASS/FAIL. */
|
|
3431
|
+
interface ScoreboardRow {
|
|
3432
|
+
storyId: string;
|
|
3433
|
+
storyTitle: string;
|
|
3434
|
+
reqId: string;
|
|
3435
|
+
reqTitle: string;
|
|
3436
|
+
status: 'PASS' | 'FAIL';
|
|
3437
|
+
evidence: string[];
|
|
3438
|
+
}
|
|
3439
|
+
/**
|
|
3440
|
+
* Flatten story verdicts into the per-requirement scoreboard — the literal
|
|
3441
|
+
* Jira tick-off: one row per (story, requirement) with PASS/FAIL and the
|
|
3442
|
+
* evidence behind the verdict.
|
|
3443
|
+
*/
|
|
3444
|
+
declare function userStoryScoreboard(verdicts: readonly UserStoryVerdict[]): ScoreboardRow[];
|
|
3445
|
+
/** Launch-readiness headline counts rolled up from the per-requirement rows. */
|
|
3446
|
+
interface ScoreboardSummary {
|
|
3447
|
+
/** Distinct user stories on the board. */
|
|
3448
|
+
stories: number;
|
|
3449
|
+
/** Stories whose every requirement passed. */
|
|
3450
|
+
storiesFullyComplete: number;
|
|
3451
|
+
/** Total (story, requirement) rows. */
|
|
3452
|
+
requirements: number;
|
|
3453
|
+
/** Rows with status PASS. */
|
|
3454
|
+
passed: number;
|
|
3455
|
+
/** Rows with status FAIL. */
|
|
3456
|
+
failed: number;
|
|
3457
|
+
/** passed / requirements; 0 when there are no rows. */
|
|
3458
|
+
passRate: number;
|
|
3459
|
+
}
|
|
3460
|
+
/** Roll the per-requirement rows up into the launch headline counts. */
|
|
3461
|
+
declare function scoreboardSummary(rows: readonly ScoreboardRow[]): ScoreboardSummary;
|
|
3462
|
+
interface ScoreboardRenderOptions {
|
|
3463
|
+
/** Document H1. Defaults to a generic playback title. */
|
|
3464
|
+
title?: string;
|
|
3465
|
+
/** Key/value run metadata rendered under the headline (runId, backend, model, date). */
|
|
3466
|
+
meta?: Record<string, string>;
|
|
3467
|
+
/** Max chars of joined evidence shown per row. Default 160. */
|
|
3468
|
+
maxEvidenceChars?: number;
|
|
3469
|
+
}
|
|
3470
|
+
/**
|
|
3471
|
+
* Render the scoreboard as a launch-readiness Markdown document — the literal
|
|
3472
|
+
* "tick off every user story" artifact: a headline roll-up, the open tickets
|
|
3473
|
+
* (FAIL rows) up top as the launch blockers, then a per-story table of
|
|
3474
|
+
* requirement → PASS/FAIL with the evidence behind each verdict. Pure: same
|
|
3475
|
+
* rows in, same bytes out (no clock/random), so it is safe to snapshot.
|
|
3476
|
+
*/
|
|
3477
|
+
declare function renderScoreboardMarkdown(rows: readonly ScoreboardRow[], opts?: ScoreboardRenderOptions): string;
|
|
3478
|
+
//#endregion
|
|
3479
|
+
//#region src/campaign/presets/segmented-profile-matrix.d.ts
|
|
3480
|
+
interface ProfileMatrixRow {
|
|
3481
|
+
rowId: string;
|
|
3482
|
+
ordinal: number;
|
|
3483
|
+
profileId: string;
|
|
3484
|
+
scenarioId: string;
|
|
3485
|
+
rep: number;
|
|
3486
|
+
}
|
|
3487
|
+
interface ProfileMatrixPlan<TScenario extends Scenario, TArtifact> {
|
|
3488
|
+
readonly schemaVersion: 1;
|
|
3489
|
+
readonly matrixId: string;
|
|
3490
|
+
readonly experimentId: string;
|
|
3491
|
+
readonly planDigest: `sha256:${string}`;
|
|
3492
|
+
readonly profiles: readonly AgentProfile$1[];
|
|
3493
|
+
readonly scenarios: readonly TScenario[];
|
|
3494
|
+
readonly judges: readonly JudgeConfig<TArtifact, TScenario>[];
|
|
3495
|
+
readonly reps: number;
|
|
3496
|
+
readonly seed: number;
|
|
3497
|
+
readonly splitTag: NonNullable<RunProfileMatrixOptions<TScenario, TArtifact>['splitTag']>;
|
|
3498
|
+
readonly commitSha: string;
|
|
3499
|
+
/** Stable dispatch implementation/configuration identity used by cell caches. */
|
|
3500
|
+
readonly dispatchRef: string;
|
|
3501
|
+
readonly integrity: NonNullable<RunProfileMatrixOptions<TScenario, TArtifact>['integrity']>;
|
|
3502
|
+
readonly allowMixed: boolean;
|
|
3503
|
+
readonly validate: boolean;
|
|
3504
|
+
readonly personaOf?: (scenario: TScenario) => string;
|
|
3505
|
+
readonly corpusText?: (artifact: TArtifact, scenario: TScenario) => {
|
|
3506
|
+
prompt: string;
|
|
3507
|
+
completion: string;
|
|
3508
|
+
} | undefined;
|
|
3509
|
+
readonly rows: readonly ProfileMatrixRow[];
|
|
3510
|
+
}
|
|
3511
|
+
interface CreateProfileMatrixPlanOptions<TScenario extends Scenario, TArtifact> extends Pick<RunProfileMatrixOptions<TScenario, TArtifact>, 'profiles' | 'scenarios' | 'judges' | 'commitSha' | 'experimentId' | 'splitTag' | 'reps' | 'seed' | 'integrity' | 'allowMixed' | 'validate' | 'personaOf' | 'corpusText'> {
|
|
3512
|
+
/** Stable dispatch implementation/configuration identity used by cell caches. */
|
|
3513
|
+
dispatchRef: string;
|
|
3514
|
+
}
|
|
3515
|
+
interface ProfileMatrixCoverage {
|
|
3516
|
+
expected: number;
|
|
3517
|
+
present: number;
|
|
3518
|
+
missing: string[];
|
|
3519
|
+
failed: string[];
|
|
3520
|
+
zeroScore: string[];
|
|
3521
|
+
}
|
|
3522
|
+
interface RunProfileMatrixSegmentOptions<TScenario extends Scenario, TArtifact> {
|
|
3523
|
+
plan: ProfileMatrixPlan<TScenario, TArtifact>;
|
|
3524
|
+
/** Stable external grant or attempt identity. Reuse it to resume. */
|
|
3525
|
+
segmentId: string;
|
|
3526
|
+
/** Explicit row ids from `plan.rows`; duplicate or unknown rows fail. */
|
|
3527
|
+
rows: readonly (string | ProfileMatrixRow)[];
|
|
3528
|
+
dispatch: ProfileDispatchFn<TScenario, TArtifact>;
|
|
3529
|
+
runDir: string;
|
|
3530
|
+
storage?: CampaignStorage;
|
|
3531
|
+
maxConcurrency?: number;
|
|
3532
|
+
maxProfileConcurrency?: number;
|
|
3533
|
+
costCeiling?: number;
|
|
3534
|
+
labeledStore?: RunProfileMatrixOptions<TScenario, TArtifact>['labeledStore'];
|
|
3535
|
+
captureSource?: RunProfileMatrixOptions<TScenario, TArtifact>['captureSource'];
|
|
3536
|
+
now?: () => Date;
|
|
3537
|
+
}
|
|
3538
|
+
interface ProfileMatrixSegmentResult<TArtifact, TScenario extends Scenario> {
|
|
3539
|
+
segmentId: string;
|
|
3540
|
+
rowIds: string[];
|
|
3541
|
+
matrix: RunProfileMatrixResult<TArtifact, TScenario>;
|
|
3542
|
+
coverage: ProfileMatrixCoverage;
|
|
3543
|
+
}
|
|
3544
|
+
interface FinalizeProfileMatrixOptions<TScenario extends Scenario, TArtifact> {
|
|
3545
|
+
plan: ProfileMatrixPlan<TScenario, TArtifact>;
|
|
3546
|
+
runDir: string;
|
|
3547
|
+
storage?: CampaignStorage;
|
|
3548
|
+
maxConcurrency?: number;
|
|
3549
|
+
maxProfileConcurrency?: number;
|
|
3550
|
+
now?: () => Date;
|
|
3551
|
+
}
|
|
3552
|
+
interface FinalizedProfileMatrixResult<TArtifact, TScenario extends Scenario> extends RunProfileMatrixResult<TArtifact, TScenario> {
|
|
3553
|
+
coverage: ProfileMatrixCoverage;
|
|
3554
|
+
}
|
|
3555
|
+
declare function createProfileMatrixPlan<TScenario extends Scenario, TArtifact>(opts: CreateProfileMatrixPlanOptions<TScenario, TArtifact>): ProfileMatrixPlan<TScenario, TArtifact>;
|
|
3556
|
+
declare function runProfileMatrixSegment<TScenario extends Scenario, TArtifact>(opts: RunProfileMatrixSegmentOptions<TScenario, TArtifact>): Promise<ProfileMatrixSegmentResult<TArtifact, TScenario>>;
|
|
3557
|
+
declare function finalizeProfileMatrix<TScenario extends Scenario, TArtifact>(opts: FinalizeProfileMatrixOptions<TScenario, TArtifact>): Promise<FinalizedProfileMatrixResult<TArtifact, TScenario>>;
|
|
3558
|
+
//#endregion
|
|
3559
|
+
//#region src/campaign/run-dir.d.ts
|
|
3560
|
+
/** The shared, out-of-repo root for campaign/benchmark run bundles. Keeping run
|
|
3561
|
+
* outputs here means they never land in a repo working tree (no per-repo
|
|
3562
|
+
* gitignore, no clutter, no accidental commits). Layout:
|
|
3563
|
+
* ~/.tangle/traces/<repo>/runs/<runName>/
|
|
3564
|
+
* where <repo> disambiguates runs across repos in one place. */
|
|
3565
|
+
declare function tangleTracesRoot(): string;
|
|
3566
|
+
/** Resolve a campaign `runDir`. An absolute path is honored as-is (the caller
|
|
3567
|
+
* chose an explicit location). A bare name is placed under the shared home root
|
|
3568
|
+
* so bundles never pollute a repo working tree — the default the harness should
|
|
3569
|
+
* compute so callers pass a *name*, not a path. */
|
|
3570
|
+
declare function resolveRunDir(runDir: string, repo?: string): string;
|
|
3571
|
+
//#endregion
|
|
3572
|
+
//#region src/campaign/scenario-selection.d.ts
|
|
3573
|
+
/**
|
|
3574
|
+
* Discriminative scenario selection (research claim E2).
|
|
3575
|
+
*
|
|
3576
|
+
* The OR benchmark is SATURATING: run 7 measured ~75% tied holdout cells — most
|
|
3577
|
+
* problems are solved optimally by the baseline AND every candidate, so those
|
|
3578
|
+
* paired cells carry zero signal. A random/balanced holdout split spends its
|
|
3579
|
+
* budget on scenarios that cannot separate candidates.
|
|
3580
|
+
*
|
|
3581
|
+
* This picks the holdout by DISCRIMINATION power instead: a scenario every
|
|
3582
|
+
* candidate scores identically (variance ~0) carries no signal; one where the
|
|
3583
|
+
* scores spread carries the most. We drop fully saturated ties so each paired
|
|
3584
|
+
* holdout cell is spent on a scenario that can actually move a verdict.
|
|
3585
|
+
*/
|
|
3586
|
+
/** Per-scenario observation: the composite scores each candidate earned on it. */
|
|
3587
|
+
interface ScenarioSignal {
|
|
3588
|
+
scenarioId: string;
|
|
3589
|
+
/** Per-candidate composite scores observed for this scenario (>=1 values). */
|
|
3590
|
+
scores: number[];
|
|
3591
|
+
}
|
|
3592
|
+
interface DiscriminationScore {
|
|
3593
|
+
scenarioId: string;
|
|
3594
|
+
/** Higher = separates candidates more (spread of their scores). */
|
|
3595
|
+
discrimination: number;
|
|
3596
|
+
/** Higher = easier / more-saturated (mean of candidate scores). */
|
|
3597
|
+
meanScore: number;
|
|
3598
|
+
variance: number;
|
|
3599
|
+
/** variance ~0 AND meanScore at/above the ceiling ⇒ a saturated tie, no signal. */
|
|
3600
|
+
tied: boolean;
|
|
3601
|
+
}
|
|
3602
|
+
/**
|
|
3603
|
+
* Rank scenarios by how well they DISCRIMINATE candidates.
|
|
3604
|
+
*
|
|
3605
|
+
* `discrimination = variance` (spread of the candidate scores) — kept simple on
|
|
3606
|
+
* purpose; the headroom term (`saturationCeiling - meanScore`) only breaks ties
|
|
3607
|
+
* so that, among equally spread scenarios, the one with more room to improve
|
|
3608
|
+
* ranks first. Returned sorted by the deterministic order above.
|
|
3609
|
+
*/
|
|
3610
|
+
declare function scoreDiscrimination(signals: ScenarioSignal[], opts?: {
|
|
3611
|
+
saturationCeiling?: number;
|
|
3612
|
+
}): DiscriminationScore[];
|
|
3613
|
+
/**
|
|
3614
|
+
* Select the top-`k` most discriminative scenario ids for a holdout, EXCLUDING
|
|
3615
|
+
* fully saturated ties when enough non-tied scenarios exist (a tie in the
|
|
3616
|
+
* holdout wastes a paired cell).
|
|
3617
|
+
*
|
|
3618
|
+
* Prefers non-tied scenarios; if fewer than `k` non-tied exist, fills with the
|
|
3619
|
+
* least-saturated tied ones (tied scenarios are already ordered least-saturated
|
|
3620
|
+
* first by `meanScore` asc). Deterministic. Throws if `k < 1`. If
|
|
3621
|
+
* `signals.length <= k`, returns all ids in discrimination order.
|
|
3622
|
+
*/
|
|
3623
|
+
declare function selectDiscriminative(signals: ScenarioSignal[], k: number, opts?: {
|
|
3624
|
+
saturationCeiling?: number;
|
|
3625
|
+
}): string[];
|
|
3626
|
+
//#endregion
|
|
3627
|
+
//#region src/campaign/score-utils.d.ts
|
|
3628
|
+
/** Mean composite across cells with complete task-quality evidence.
|
|
3629
|
+
* Partial judge results remain on their cells but never enter this value.
|
|
3630
|
+
* A campaign with no complete score has no numeric mean and fails loudly. */
|
|
3631
|
+
declare function campaignMeanComposite<TArtifact, TScenario extends Scenario>(campaign: CampaignResult<TArtifact, TScenario>): number;
|
|
3632
|
+
/** Compare fixed-length lexicographic rank keys where each element is higher-is-better.
|
|
3633
|
+
* Returns a positive number when `a` ranks above `b`, negative when below, and
|
|
3634
|
+
* zero when equal. */
|
|
3635
|
+
declare function compareRankKeys(a: readonly number[], b: readonly number[]): number;
|
|
3636
|
+
interface CampaignBreakdown {
|
|
3637
|
+
/** Mean score per judge dimension across all cells. */
|
|
3638
|
+
dimensions: Record<string, number>;
|
|
3639
|
+
/** Per-scenario composite (mean over reps + judges) + the judge's free-form
|
|
3640
|
+
* `notes` for that scenario (the "why" a reflective proposer grounds on) +
|
|
3641
|
+
* an optional `emitted` excerpt of the candidate's raw output (the "what it
|
|
3642
|
+
* actually did" a reflective proposer grounds on). */
|
|
3643
|
+
scenarios: Array<{
|
|
3644
|
+
scenarioId: string;
|
|
3645
|
+
composite: number;
|
|
3646
|
+
notes?: string;
|
|
3647
|
+
emitted?: string;
|
|
3648
|
+
}>;
|
|
3649
|
+
}
|
|
3650
|
+
/** Per-candidate evidence a reflective/patch proposer grounds its next proposal
|
|
3651
|
+
* on: mean score per judge dimension + per-scenario composite. */
|
|
3652
|
+
declare function campaignBreakdown<TArtifact, TScenario extends Scenario>(campaign: CampaignResult<TArtifact, TScenario>): CampaignBreakdown;
|
|
3653
|
+
//#endregion
|
|
3654
|
+
//#region src/campaign/single-run-lock.d.ts
|
|
3655
|
+
/**
|
|
3656
|
+
* Single-run lock for evaluations that share one mutable environment.
|
|
3657
|
+
*
|
|
3658
|
+
* Two concurrent runs against a shared stateful gym silently corrupt each
|
|
3659
|
+
* other: each resets/mutates environment state mid-cell of the other, and
|
|
3660
|
+
* every score from both becomes garbage that LOOKS like worker variance
|
|
3661
|
+
* (agent-lab R357 burned hours on flip-flopping scores before tracing them
|
|
3662
|
+
* to exactly this). The fix is a pid lockfile: refuse to start while a live
|
|
3663
|
+
* holder exists, reclaim stale locks whose pid is gone, release only if the
|
|
3664
|
+
* lock is still ours.
|
|
3665
|
+
*
|
|
3666
|
+
* `alsoCheck` exists because independent runners can guard the same shared
|
|
3667
|
+
* resource with differently named lockfiles; a runner must respect all of
|
|
3668
|
+
* them even though it writes only its own.
|
|
3669
|
+
*/
|
|
3670
|
+
interface SingleRunLockOptions {
|
|
3671
|
+
/** Lockfile this runner writes (and checks). */
|
|
3672
|
+
readonly lockPath: string;
|
|
3673
|
+
/** Other runners' lockfiles guarding the same resource; checked, never written. */
|
|
3674
|
+
readonly alsoCheck?: readonly string[];
|
|
3675
|
+
/** Install a process 'exit' hook that releases the lock. Default true. */
|
|
3676
|
+
readonly releaseOnExit?: boolean;
|
|
3677
|
+
/** Owner pid recorded in the lockfile metadata. Default process.pid. */
|
|
3678
|
+
readonly pid?: number;
|
|
3679
|
+
}
|
|
3680
|
+
interface SingleRunLock {
|
|
3681
|
+
/** Remove the lockfile if this process still owns it. Idempotent. */
|
|
3682
|
+
release(): void;
|
|
3683
|
+
}
|
|
3684
|
+
/**
|
|
3685
|
+
* Acquire the lock or throw naming the live holder. A stale lock (holder pid
|
|
3686
|
+
* no longer running) is reclaimed by one contender. An interrupted reclaim
|
|
3687
|
+
* leaves a marker that fails closed instead of admitting overlapping runs.
|
|
3688
|
+
*/
|
|
3689
|
+
declare function acquireSingleRunLock(opts: SingleRunLockOptions): SingleRunLock;
|
|
3690
|
+
//#endregion
|
|
3691
|
+
//#region src/campaign/surface-identity.d.ts
|
|
3692
|
+
/** Validate the immutable identity shape; the owning executor verifies the Git objects and patch. */
|
|
3693
|
+
declare function assertCodeSurfaceIdentity(surface: unknown): asserts surface is CodeSurface;
|
|
3694
|
+
/**
|
|
3695
|
+
* Deterministic identity material for a component surface.
|
|
3696
|
+
*
|
|
3697
|
+
* `canonicalString` orders keys by UTF-16 code unit (RFC 8785), which is a
|
|
3698
|
+
* property of the value alone. The previous material ordered them with
|
|
3699
|
+
* `localeCompare`, which reads the host's collation — so the same surface
|
|
3700
|
+
* could produce two different identities on two machines, and the stored
|
|
3701
|
+
* identity would stop matching a recomputation of the identical surface.
|
|
3702
|
+
*/
|
|
3703
|
+
declare function componentSurfaceIdentityMaterial(surface: ComponentSurface): string;
|
|
3704
|
+
/** Canonical, location-independent identity of a finalized code candidate.
|
|
3705
|
+
* Commit metadata is excluded: two commits with the same base, final tree,
|
|
3706
|
+
* and patch bytes are the same executable candidate. */
|
|
3707
|
+
declare function codeSurfaceIdentityMaterial(surface: CodeSurface): string;
|
|
3708
|
+
/** Full SHA-256 content identity for a prompt or finalized code surface. */
|
|
3709
|
+
declare function surfaceContentHash(surface: MutableSurface): `sha256:${string}`;
|
|
3710
|
+
/** Short loop key derived from the same content identity as provenance. */
|
|
3711
|
+
declare function surfaceHash(surface: MutableSurface): string;
|
|
3712
|
+
/** Canonical customer-visible description of the exact before/after surfaces. */
|
|
3713
|
+
declare function renderSurfaceDiff(winnerSurface: MutableSurface, baselineSurface: MutableSurface): string;
|
|
3714
|
+
/** Bind a campaign cache entry to the exact surface and caller-owned execution revision. */
|
|
3715
|
+
declare function surfaceDispatchRef(surface: MutableSurface, executionRef?: string): string;
|
|
3716
|
+
//#endregion
|
|
3717
|
+
//#region src/campaign/upstream-evaluators.d.ts
|
|
3718
|
+
interface PhoenixEvaluationResultLike {
|
|
3719
|
+
score?: number;
|
|
3720
|
+
label?: string;
|
|
3721
|
+
explanation?: string;
|
|
3722
|
+
}
|
|
3723
|
+
interface PhoenixEvaluatorLike<TRecord extends Record<string, unknown>> {
|
|
3724
|
+
name: string;
|
|
3725
|
+
kind: 'LLM' | 'CODE';
|
|
3726
|
+
optimizationDirection?: 'MAXIMIZE' | 'MINIMIZE' | 'NEUTRAL';
|
|
3727
|
+
evaluate(record: TRecord, context: UpstreamEvaluationContext): Promise<PhoenixEvaluationResultLike>;
|
|
3728
|
+
}
|
|
3729
|
+
interface AutoevalsScoreLike {
|
|
3730
|
+
name: string;
|
|
3731
|
+
score: number | null;
|
|
3732
|
+
metadata?: Record<string, unknown>;
|
|
3733
|
+
}
|
|
3734
|
+
type AutoevalsScorerLike<TInput extends Record<string, unknown>> = (input: TInput, context: UpstreamEvaluationContext) => AutoevalsScoreLike | Promise<AutoevalsScoreLike>;
|
|
3735
|
+
interface UpstreamEvaluationContext {
|
|
3736
|
+
readonly signal: AbortSignal;
|
|
3737
|
+
readonly callId?: string;
|
|
3738
|
+
}
|
|
3739
|
+
type PaidEvaluationOptions<TResult> = Pick<RunPaidCallInput<TResult>, 'maximumCharge' | 'receipt' | 'receiptFromError'> & {
|
|
3740
|
+
model: string;
|
|
3741
|
+
};
|
|
3742
|
+
interface UpstreamJudgeOptions<TScenario extends Scenario> {
|
|
3743
|
+
name?: string;
|
|
3744
|
+
dimension?: string;
|
|
3745
|
+
judgeVersion?: string;
|
|
3746
|
+
appliesTo?: (scenario: TScenario) => boolean;
|
|
3747
|
+
/** Convert the upstream score when its native scale is not higher-is-better. */
|
|
3748
|
+
toComposite?: (score: number) => number;
|
|
3749
|
+
}
|
|
3750
|
+
declare function phoenixEvaluatorJudge<TRecord extends Record<string, unknown>, TArtifact, TScenario extends Scenario = Scenario>(evaluator: PhoenixEvaluatorLike<TRecord>, options: UpstreamJudgeOptions<TScenario> & {
|
|
3751
|
+
mapInput(input: {
|
|
3752
|
+
artifact: TArtifact;
|
|
3753
|
+
scenario: TScenario;
|
|
3754
|
+
}): TRecord;
|
|
3755
|
+
paidCall?: PaidEvaluationOptions<PhoenixEvaluationResultLike>;
|
|
3756
|
+
}): JudgeConfig<TArtifact, TScenario>;
|
|
3757
|
+
declare function autoevalsScorerJudge<TInput extends Record<string, unknown>, TArtifact, TScenario extends Scenario = Scenario>(scorer: AutoevalsScorerLike<TInput>, options: UpstreamJudgeOptions<TScenario> & {
|
|
3758
|
+
name: string;
|
|
3759
|
+
mapInput(input: {
|
|
3760
|
+
artifact: TArtifact;
|
|
3761
|
+
scenario: TScenario;
|
|
3762
|
+
}): TInput;
|
|
3763
|
+
} & ({
|
|
3764
|
+
kind: 'CODE';
|
|
3765
|
+
paidCall?: never;
|
|
3766
|
+
} | {
|
|
3767
|
+
kind: 'LLM';
|
|
3768
|
+
paidCall: PaidEvaluationOptions<AutoevalsScoreLike>;
|
|
3769
|
+
})): JudgeConfig<TArtifact, TScenario>;
|
|
3770
|
+
//#endregion
|
|
3771
|
+
//#region src/campaign/worktree/index.d.ts
|
|
3772
|
+
type GitOutput = string | Uint8Array;
|
|
3773
|
+
type GitEnvironment = Readonly<Record<string, string>>;
|
|
3774
|
+
type GitRunner = (args: string[], cwd: string, env?: GitEnvironment) => GitOutput;
|
|
3775
|
+
interface Worktree {
|
|
3776
|
+
/** Absolute path to the checked-out worktree directory. */
|
|
3777
|
+
readonly path: string;
|
|
3778
|
+
/** The branch the worktree is on (becomes the PR branch on promotion). */
|
|
3779
|
+
readonly branch: string;
|
|
3780
|
+
/** The ref the worktree was forked from. */
|
|
3781
|
+
readonly baseRef: string;
|
|
3782
|
+
/** Exact commit `baseRef` resolved to before the worktree was created. */
|
|
3783
|
+
readonly baseCommit: string;
|
|
3784
|
+
/** Exact tree object for `baseCommit`. */
|
|
3785
|
+
readonly baseTree: string;
|
|
3786
|
+
}
|
|
3787
|
+
interface WorktreeAdapter {
|
|
3788
|
+
/** Create an isolated worktree on a fresh branch off `baseRef`. */
|
|
3789
|
+
create(opts: {
|
|
3790
|
+
baseRef: string;
|
|
3791
|
+
label: string;
|
|
3792
|
+
}): Promise<Worktree>;
|
|
3793
|
+
/** Commit pending changes, freeze the exact Git objects + binary patch, and
|
|
3794
|
+
* verify the worktree still matches that identity. */
|
|
3795
|
+
finalize(worktree: Worktree, summary: string): Promise<CodeSurface>;
|
|
3796
|
+
/** Idempotently remove the worktree and branch. Safe to retry after partial cleanup. */
|
|
3797
|
+
discard(worktree: Worktree): Promise<void>;
|
|
3798
|
+
}
|
|
3799
|
+
/** Typed failure from a `WorktreeAdapter` operation (create/finalize/discard) — wraps the underlying git error as `cause`. */
|
|
3800
|
+
declare class WorktreeAdapterError extends Error {
|
|
3801
|
+
readonly cause?: unknown;
|
|
3802
|
+
constructor(message: string, cause?: unknown);
|
|
3803
|
+
}
|
|
3804
|
+
interface GitWorktreeAdapterOptions {
|
|
3805
|
+
/** Repo root the worktrees fork from. */
|
|
3806
|
+
repoRoot: string;
|
|
3807
|
+
/** Directory worktrees are created under. Default: `<repoRoot>/.worktrees`. */
|
|
3808
|
+
worktreeDir?: string;
|
|
3809
|
+
/** Branch-name prefix. Default: `improve`. */
|
|
3810
|
+
branchPrefix?: string;
|
|
3811
|
+
/** Test seam — defaults to a real `git` runner. The return value must contain
|
|
3812
|
+
* stdout verbatim, and runners that execute Git must forward the optional
|
|
3813
|
+
* environment overrides used to isolate patch generation. */
|
|
3814
|
+
git?: GitRunner;
|
|
3815
|
+
}
|
|
3816
|
+
interface CodeSurfaceVerification {
|
|
3817
|
+
/** Verified worktree path. */
|
|
3818
|
+
path: string;
|
|
3819
|
+
/** Git's canonical root for the verified checkout. */
|
|
3820
|
+
repoRoot: string;
|
|
3821
|
+
/** Recomputed full content identity. */
|
|
3822
|
+
contentHash: `sha256:${string}`;
|
|
3823
|
+
/** Exact verified binary-patch bytes. Candidate-bundle builders encode this
|
|
3824
|
+
* directly instead of reproducing Git diff options. */
|
|
3825
|
+
patchBytes: Uint8Array;
|
|
3826
|
+
}
|
|
3827
|
+
/**
|
|
3828
|
+
* Git-backed `WorktreeAdapter`: creates isolated worktrees on fresh branches, commits agent changes, and discards losers.
|
|
3829
|
+
*/
|
|
3830
|
+
declare function gitWorktreeAdapter(opts: GitWorktreeAdapterOptions): WorktreeAdapter;
|
|
3831
|
+
/** Verify a finalized code surface against its current checkout. This rejects
|
|
3832
|
+
* dirty/ignored files, moved refs, missing Git objects, raw byte/mode
|
|
3833
|
+
* mismatches, external symlinks, and submodules. */
|
|
3834
|
+
declare function verifyCodeSurface(surface: CodeSurface, worktreeDir?: string): CodeSurfaceVerification;
|
|
3835
|
+
/** Resolve a code candidate for evaluation only after verifying its immutable
|
|
3836
|
+
* identity against the checkout at `worktreeRef`. */
|
|
3837
|
+
declare function resolveWorktreePath(surface: CodeSurface, worktreeDir?: string): string;
|
|
3838
|
+
//#endregion
|
|
3839
|
+
export { UserStoryVerdict as $, OpenAutoPrOptions as $a, SearchHistoryCoverageRow as $i, campaignMeasurementDigest as $n, OptimizationMethodProvenance as $r, PlanEvalFixtureRunOptions as $t, ScenarioSignal as A, SearchOperationKind as Aa, RunImprovementLoopOptions as Ai, CrossSurfacePairEvidence as An, SkillOptTrainerConfig as Ar, completionVerdict as At, ProfileMatrixRow as B, SearchSurfaceKind as Ba, SearchLedgerBinding as Bi, TraceAnalystScenario as Bn, OptimizerModelBudget as Br, ScoredRollout as Bt, SingleRunLockOptions as C, SearchLedgerAppendResult as Ca, LlmJudgeDimension as Ci, CrossSurfaceInteractionAwareSelection as Cn, assertCampaignSplitIdentity as Co, CanaryKind as Cr, CorrectnessChecker as Ct, campaignMeanComposite as D, SearchLedgerReplay as Da, isTransientTransportFailure as Di, CrossSurfaceInteractionTask as Dn, composeGate as Dr, RequirementCheck as Dt, campaignBreakdown as E, SearchLedgerHash as Ea, TransientFailureOptions as Ei, CrossSurfaceInteractionReport as En, campaignSplitDigestFromIdentities as Eo, runCanaries as Er, ProducedState as Et, CreateProfileMatrixPlanOptions as F, SearchPlannedOperation as Fa, RunOptimizationResult as Fi, CrossSurfaceSelectionPolicy as Fn, GepaOptimizationMethodConfig as Fr, FsLabeledScenarioStoreOptions as Ft, runProfileMatrixSegment as G, validateSearchLedgerEvent as Ga, GepaCandidatePopulationArtifact as Gi, EmitLoopProvenanceResult as Gn, ExternalOptimizationExample as Gr, neutralizationGate as Gt, RunProfileMatrixSegmentOptions as H, SearchTaskOutcome as Ha, SearchRecorderOptions as Hi, traceAnalystQualityJudge as Hn, ExternalTextOptimizationMethodConfig as Hr, classifyUngroundedLiterals as Ht, FinalizeProfileMatrixOptions as I, SearchPlannedTask as Ia, runOptimization as Ii, CrossSurfaceSelections as In, GepaOptimizationRecipe as Ir, LabeledScenarioStoreError as It, PlaybackStep as J, SearchLedgerIntegrityError as Ja, GepaCandidateSelectionScore as Ji, LoopProvenanceCandidate as Jn, CompareOptimizationMethodsOptions as Jr, EvalFixtureLoadOptions as Jt, PlaybackContext as K, SearchLedgerConflictError as Ka, GepaCandidatePopulationCandidate as Ki, LoopProvenanceArgsFromResult as Kn, ExternalTextEvaluationResponse as Kr, EvalFixture as Kt, FinalizedProfileMatrixResult as L, SearchSourceRef as La, MeasuredSearchCandidate as Li, CrossSurfaceTaskRow as Ln, GepaRunnerCommand as Lr, RolloutArgumentDiff as Lt, selectDiscriminative as M, SearchPlan as Ma, runImprovementLoop as Mi, CrossSurfacePairwiseEntry as Mn, GepaAdaptiveEngineRun as Mr, verifyCompletion as Mt, resolveRunDir as N, SearchPlanExtendedEvent as Na, PremeasuredOptimizationBaseline as Ni, CrossSurfaceRankedSingle as Nn, GepaEngineOptions as Nr, neutralizeText as Nt, compareRankKeys as O, SearchLedgerTrustedHeadMode as Oa, quotaExhaustedUntil as Oi, CrossSurfaceNaiveStackSelection as On, SkillOptOptimizationMethodConfig as Or, SatisfiedBy as Ot, tangleTracesRoot as P, SearchPlannedEvent as Pa, RunOptimizationOptions as Pi, CrossSurfaceRelativeCost as Pn, GepaEngineRun as Pr, FsLabeledScenarioStore as Pt, UserStory as Q, crowdedFrontierParent as Qa, SearchHistoryCoverage as Qi, buildLoopProvenanceRecord as Qn, OptimizationMethodPairwise as Qr, LoadEvalFixtureScenariosOptions as Qt, ProfileMatrixCoverage as R, SearchSurfaceEffect as Ra, ProposedSearchCandidate as Ri, BuildTraceAnalystSurfaceDispatchOptions as Rn, gepaOptimizationMethod as Rr, RolloutArgumentDiffOptions as Rt, SingleRunLock as S, SearchLedger as Sa, runReferenceEquivalenceJudge as Si, CrossSurfaceIneligibilityReason as Sn, assertCampaignDesign as So, CanaryEvaluation as Sr, CompletionVerdict as St, CampaignBreakdown as T, SearchLedgerEvent as Ta, llmJudge as Ti, CrossSurfaceInteractionPath as Tn, campaignSplitDigest as To, CanaryReport as Tr, ProducedProposal as Tt, createProfileMatrixPlan as U, SearchTokenAccounting as Ua, SearchRunIdentity as Ui, BuildLoopProvenanceArgs as Un, ExternalTextOptimizerContext as Ur, rolloutArgumentDiff as Ut, ProfileMatrixSegmentResult as V, SearchTaskAttemptedEvent as Va, SearchRecorder as Vi, buildTraceAnalystSurfaceDispatch as Vn, externalTextOptimizationMethod as Vr, UngroundedLiteralReport as Vt, finalizeProfileMatrix as W, openSearchLedger as Wa, recordCandidatePopulationSearch as Wi, EmitLoopProvenanceArgs as Wn, ExternalTextOptimizerResult as Wr, NeutralizationGateOptions as Wt, ScoreboardRow as X, ParentSelectionContext as Xa, CreateSearchHistoryReceiptInput as Xi, LoopProvenanceOptimizationMethod as Xn, OptimizationMethodComparison as Xr, EvalFixtureScenario as Xt, ScoreboardRenderOptions as Y, CrowdedFrontierParentOptions as Ya, readGepaCandidatePopulationArtifact as Yi, LoopProvenanceEvidence as Yn, OptimizationMethod as Yr, EvalFixtureRunPlan as Yt, ScoreboardSummary as Z, ParentSelector as Za, SearchHistoryAuditSummary as Zi, LoopProvenanceRecord as Zn, OptimizationMethodInput as Zr, EvalFixtureValidationMode as Zt, componentSurfaceIdentityMaterial as _, SearchCandidateSlotClosedEvent as _a, ReferenceEquivalenceJudgeInput as _i, CrossSurfaceComponentEvidence as _n, cellCachePath as _o, RedTeamReport as _r, ProposalEventLike as _t, WorktreeAdapterError as a, createSearchHistoryReceipt as aa, compareOptimizationMethods as ai, AnalyzeCrossSurfaceInteractionsInput as an, CampaignCellRetryPolicy as ao, provenanceSpansPath as ar, ProfileDispatchFn as at, surfaceDispatchRef as b, SearchCostAccounting as ba, ReferenceEquivalenceScenario as bi, CrossSurfaceEligibility as bn, fsCampaignStorage as bo, scoreRedTeamOutput as br, extractProducedState as bt, verifyCodeSurface as c, FileSearchLedger as ca, combineComparisonCosts as ci, CrossSurfaceAttemptCompleteness as cn, CampaignRunPlan as co, heldOutGate as cr, RunProfileMatrixOptions as ct, PhoenixEvaluationResultLike as d, SearchArtifactRef as da, ExternalOptimizerObservationArtifact as di, CrossSurfaceCandidate as dn, planCampaignRun as do, DefaultProductionRewardHackingOptions as dr, runProfileMatrix as dt, SearchHistoryPolicy as ea, OptimizationMethodResult as ei, discoverEvalFixtures as en, OpenAutoPrResult as eo, canonicalDigest as er, makePlaybackDispatch as et, PhoenixEvaluatorLike as f, SearchAttemptAccounting as fa, ExternalOptimizerObservationSummary as fi, CrossSurfaceCandidateComparison as fn, CacheIssueReason as fo, defaultProductionGate as fr, BackendIntegrityError as ft, codeSurfaceIdentityMaterial as g, SearchCandidateSlot as ga, REFERENCE_EQUIVALENCE_JUDGE_VERSION as gi, CrossSurfaceComponent as gn, buildCellSchedule as go, RedTeamFinding as gr, ArtifactEventLike as gt, assertCodeSurfaceIdentity as h, SearchCandidateRegisteredEvent as ha, REFERENCE_EQUIVALENCE_INPUT_LIMITS as hi, CrossSurfaceCandidateSummary as hn, CellScheduleSlot as ho, RedTeamCategory as hr, summarizeBackendIntegrity as ht, WorktreeAdapter as i, assertSearchHistoryMatchesReplay as ia, OptimizationTokenUsage as ii, analyzeCrossSurfaceInteractions as in, CampaignCellFailureReceipt as io, provenanceRecordPath as ir, userStoryScoreboard as it, scoreDiscrimination as j, SearchOperationRecordedEvent as ja, RunImprovementLoopResult as ji, CrossSurfacePairIncompatibilityReason as jn, skillOptOptimizationMethod as jr, createLlmCorrectnessChecker as jt, DiscriminationScore as k, SearchModelIdentity as ka, transientDispatchFailure as ki, CrossSurfacePairCompatibility as kn, SkillOptRunnerCommand as kr, TaskGold as kt, AutoevalsScoreLike as l, OpenSearchLedgerOptions as la, costFromLedgerSummary as li, CrossSurfaceBestSingleSelection as ln, CampaignRunPlanCell as lo, DefaultProductionGateCheck as lr, RunProfileMatrixResult as lt, phoenixEvaluatorJudge as m, SearchCandidateLineage as ma, readExternalOptimizerObservationArtifact as mi, CrossSurfaceCandidateOutcome as mn, readCachedCell as mo, RedTeamCase as mr, assertRealBackend as mt, GitWorktreeAdapterOptions as n, SearchHistoryRequiredError as na, OptimizationMethodScore as ni, loadEvalFixtureScenarios as nn, RunEvalOptions as no, loopProvenanceArgsFromResult as nr, scoreUserStory as nt, gitWorktreeAdapter as o, searchHistoryCoverageRow as oa, optimizationTokenUsageFromSummary as oi, CrossSurfaceAdditionDecision as on, RunCampaignOptions as oo, verifyLoopProvenanceRecord as or, ProfileMatrixError as ot, autoevalsScorerJudge as p, SearchCandidateDecidedEvent as pa, ExternalOptimizerSubmittedCandidate as pi, CrossSurfaceCandidateEvidence as pn, CacheRead as po, DEFAULT_RED_TEAM_CORPUS as pr, BackendIntegrityReport as pt, PlaybackDriver as q, SearchLedgerError as qa, GepaCandidatePopulationSummary as qi, LoopProvenanceBackend as qn, decodeExternalTextCandidate as qr, EvalFixtureFile as qt, Worktree as r, assertCompleteSearchHistory as ra, OptimizationPackageSource as ri, planEvalFixtureRun as rn, runEval as ro, loopProvenanceSpans as rr, scoreboardSummary as rt, resolveWorktreePath as s, verifySearchHistoryReceipt as sa, ComparisonCost as si, CrossSurfaceAdditionRejectionReason as sn, runCampaign as so, HeldOutGateOptions as sr, ProfileSummary as st, CodeSurfaceVerification as t, SearchHistoryReceipt as ta, OptimizationMethodRunOptions as ti, loadEvalFixture as tn, openAutoPr as to, emitLoopProvenance as tr, renderScoreboardMarkdown as tt, AutoevalsScorerLike as u, SearchAccountingAudit as ua, ExternalOptimizerExecutionSummary as ui, CrossSurfaceBootstrapPolicy as un, PlanCampaignRunOptions as uo, DefaultProductionGateOptions as ur, ScenarioRollup as ut, renderSurfaceDiff as v, SearchCandidateSurface as va, ReferenceEquivalenceJudgeOptions as vi, CrossSurfaceCompositionStep as vn, CampaignStorage as vo, redTeamDataset as vr, RuntimeEventLike as vt, acquireSingleRunLock as w, SearchLedgerEntry as wa, LlmJudgeOptions as wi, CrossSurfaceInteractionEffect as wn, campaignScenarioIdentity as wo, CanaryOptions as wr, LlmCorrectnessCheckerOpts as wt, surfaceHash as x, SearchFailureReason as xa, createReferenceEquivalenceJudge as xi, CrossSurfaceEvidenceBreakdown as xn, inMemoryCampaignStorage as xo, CanaryAlert as xr, CompletionRequirement as xt, surfaceContentHash as y, SearchCompletedEvent as ya, ReferenceEquivalenceJudgeResult as yi, CrossSurfaceDistribution as yn, createRunCostLedger as yo, redTeamReport as yr, ToolCallEventLike as yt, ProfileMatrixPlan as z, SearchSurfaceEvidence as za, SearchExecutionIdentity as zi, TraceAnalystArtifact as zn, OpenAICompatibleOptimizerModel as zr, RolloutCall as zt };
|
|
3840
|
+
//# sourceMappingURL=index-DKXuBPXf.d.ts.map
|