@tangle-network/agent-eval 0.175.0 → 0.177.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +30 -0
- package/dist/adapters/http.d.ts +1 -1
- package/dist/agent-profile-cell-0gSi5ffD.js +374 -0
- package/dist/agent-profile-cell-0gSi5ffD.js.map +1 -0
- package/dist/analyst/index.d.ts +2 -2
- package/dist/analyst/index.js +3 -3
- package/dist/{benchmark-command-D_5xG9LG.js → benchmark-command-BrsZhMSk.js} +7 -7
- package/dist/{benchmark-command-D_5xG9LG.js.map → benchmark-command-BrsZhMSk.js.map} +1 -1
- package/dist/benchmarks/index.d.ts +1 -1
- package/dist/benchmarks/index.js +3 -3
- package/dist/campaign/index.d.ts +3 -3
- package/dist/campaign/index.js +9 -9
- package/dist/{campaign-BzMSCejE.js → campaign-CIn-ErlJ.js} +111 -13
- package/dist/campaign-CIn-ErlJ.js.map +1 -0
- package/dist/campaign-evidence-D8DBLqLI.js +2083 -0
- package/dist/campaign-evidence-D8DBLqLI.js.map +1 -0
- package/dist/cli.js +1 -1
- package/dist/contract/index.d.ts +4 -4
- package/dist/contract/index.js +8 -8
- package/dist/{define-agent-eval-ox5McL6e.js → define-agent-eval-CJG7LD9M.js} +54 -36
- package/dist/define-agent-eval-CJG7LD9M.js.map +1 -0
- package/dist/{define-agent-eval-V1jQyCDR.d.ts → define-agent-eval-CqmlXUfQ.d.ts} +11 -4
- package/dist/define-agent-eval-CqmlXUfQ.d.ts.map +1 -0
- package/dist/{dspy-rlm-engine-Caz2pl4L.js → dspy-rlm-engine-DqjER2sV.js} +2 -2
- package/dist/{dspy-rlm-engine-Caz2pl4L.js.map → dspy-rlm-engine-DqjER2sV.js.map} +1 -1
- package/dist/{eval-campaign-BeAjdhzC.js → eval-campaign-Cs-7MiCs.js} +4 -5
- package/dist/{eval-campaign-BeAjdhzC.js.map → eval-campaign-Cs-7MiCs.js.map} +1 -1
- package/dist/experiment/index.d.ts +3 -68
- package/dist/experiment/index.d.ts.map +1 -1
- package/dist/experiment/index.js +6 -128
- package/dist/experiment/index.js.map +1 -1
- package/dist/{attestation-XSUpbc4o.js → experiment-tracker-BKEumQug.js} +2 -96
- package/dist/experiment-tracker-BKEumQug.js.map +1 -0
- package/dist/{attestation-c1QvaBdX.d.ts → experiment-tracker-CNwqCZFD.d.ts} +2 -78
- package/dist/experiment-tracker-CNwqCZFD.d.ts.map +1 -0
- package/dist/{external-optimizer-process-CxnFL1hd.js → external-optimizer-process-Dlz8YxrT.js} +3 -3
- package/dist/{external-optimizer-process-CxnFL1hd.js.map → external-optimizer-process-Dlz8YxrT.js.map} +1 -1
- package/dist/{external-optimizer-subprocess-CQi27uEI.js → external-optimizer-subprocess-q3VzlGAO.js} +2 -2
- package/dist/{external-optimizer-subprocess-CQi27uEI.js.map → external-optimizer-subprocess-q3VzlGAO.js.map} +1 -1
- package/dist/{index-DKXuBPXf.d.ts → index-CtGf6e6X.d.ts} +50 -10
- package/dist/index-CtGf6e6X.d.ts.map +1 -0
- package/dist/{index-BTrx5s8m.d.ts → index-DuaNwvse.d.ts} +4 -4
- package/dist/{index-BTrx5s8m.d.ts.map → index-DuaNwvse.d.ts.map} +1 -1
- package/dist/{index-D-UdhAmg.d.ts → index-u0d1Jp4F.d.ts} +4 -2
- package/dist/{index-D-UdhAmg.d.ts.map → index-u0d1Jp4F.d.ts.map} +1 -1
- package/dist/index.d.ts +4 -4
- package/dist/index.js +14 -15
- package/dist/index.js.map +1 -1
- package/dist/ledger-core/index.d.ts +2 -2
- package/dist/ledger-core/index.js +2 -2
- package/dist/{ledger-core-PIfjCbKn.js → ledger-core-Cs9f7385.js} +60 -47
- package/dist/{ledger-core-PIfjCbKn.js.map → ledger-core-Cs9f7385.js.map} +1 -1
- package/dist/{llm-judge-DmNaBrXB.js → llm-judge-CV80fkYA.js} +1039 -1517
- package/dist/llm-judge-CV80fkYA.js.map +1 -0
- package/dist/{mint-vWOdD8Ae.js → mint-Cc1_zwRQ.js} +2 -2
- package/dist/{mint-vWOdD8Ae.js.map → mint-Cc1_zwRQ.js.map} +1 -1
- package/dist/openapi.json +1 -1
- package/dist/{produced-state-B8mw6zj9.js → produced-state-CQFi465p.js} +3 -2
- package/dist/{produced-state-B8mw6zj9.js.map → produced-state-CQFi465p.js.map} +1 -1
- package/dist/profile-cell.js +1 -268
- package/dist/{promotion-policy-LY9mVQ7W.js → promotion-policy-DWOm70gx.js} +2 -2
- package/dist/{promotion-policy-LY9mVQ7W.js.map → promotion-policy-DWOm70gx.js.map} +1 -1
- package/dist/{release-confidence-BsGEg_xg.js → release-confidence-BcGCclTB.js} +2 -2
- package/dist/{release-confidence-BsGEg_xg.js.map → release-confidence-BcGCclTB.js.map} +1 -1
- package/dist/reporting.js +2 -2
- package/dist/{reward-hacking-CKW4teig.js → reward-hacking-D0XwhVWE.js} +2 -215
- package/dist/reward-hacking-D0XwhVWE.js.map +1 -0
- package/dist/rl.js +5 -4
- package/dist/rl.js.map +1 -1
- package/dist/rollout/index.js +2 -2
- package/dist/{rollout-CGlDq1GI.js → rollout-DmoJVqrF.js} +2 -2
- package/dist/{rollout-CGlDq1GI.js.map → rollout-DmoJVqrF.js.map} +1 -1
- package/dist/run-record-CR63CpHK.js +216 -0
- package/dist/run-record-CR63CpHK.js.map +1 -0
- package/dist/{run-record-ZIsR9Fif.js → run-record-DQpSf7t-.js} +2 -2
- package/dist/{run-record-ZIsR9Fif.js.map → run-record-DQpSf7t-.js.map} +1 -1
- package/dist/{semantic-concept-judge-E3s_fEjB.js → semantic-concept-judge-Dw-f7TEs.js} +3 -3
- package/dist/{semantic-concept-judge-E3s_fEjB.js.map → semantic-concept-judge-Dw-f7TEs.js.map} +1 -1
- package/dist/{sequential-B51qAYE4.js → sequential-B5gXgcyp.js} +3 -3
- package/dist/{sequential-B51qAYE4.js.map → sequential-B5gXgcyp.js.map} +1 -1
- package/dist/{skillopt-optimization-method-f7399oGb.js → skillopt-optimization-method-D0o2c2yM.js} +6 -749
- package/dist/skillopt-optimization-method-D0o2c2yM.js.map +1 -0
- package/dist/{statistical-heldout-Cqb73yE9.d.ts → statistical-heldout-Z9NROFFS.d.ts} +156 -3
- package/dist/statistical-heldout-Z9NROFFS.d.ts.map +1 -0
- package/dist/{summary-report-Bgh8CpNK.js → summary-report-B16xy9Kd.js} +2 -2
- package/dist/{summary-report-Bgh8CpNK.js.map → summary-report-B16xy9Kd.js.map} +1 -1
- package/dist/traces.js +1 -1
- package/docs/campaign-proposers.md +44 -0
- package/docs/public-api.md +62 -39
- package/docs/search-history-receipts.md +48 -1
- package/package.json +1 -1
- package/dist/attestation-XSUpbc4o.js.map +0 -1
- package/dist/attestation-c1QvaBdX.d.ts.map +0 -1
- package/dist/campaign-BzMSCejE.js.map +0 -1
- package/dist/define-agent-eval-V1jQyCDR.d.ts.map +0 -1
- package/dist/define-agent-eval-ox5McL6e.js.map +0 -1
- package/dist/index-DKXuBPXf.d.ts.map +0 -1
- package/dist/llm-judge-DmNaBrXB.js.map +0 -1
- package/dist/power-preflight-CFXm0Vjo.js +0 -502
- package/dist/power-preflight-CFXm0Vjo.js.map +0 -1
- package/dist/pre-registration-D94b7Of5.js +0 -110
- package/dist/pre-registration-D94b7Of5.js.map +0 -1
- package/dist/profile-cell.js.map +0 -1
- package/dist/reward-hacking-CKW4teig.js.map +0 -1
- package/dist/skillopt-optimization-method-f7399oGb.js.map +0 -1
- package/dist/statistical-heldout-Cqb73yE9.d.ts.map +0 -1
|
@@ -1,8 +1,8 @@
|
|
|
1
1
|
import { c as CostLedgerHandle, f as CostLedgerSummary, m as CostReceipt } from "./cost-ledger-DbQdN3nO.js";
|
|
2
2
|
import { _ as ProposalFinding } from "./types-DN2WdT5S.js";
|
|
3
|
-
import { B as PowerPreflight } from "./statistical-heldout-
|
|
3
|
+
import { B as PowerPreflight, ft as EvidenceReceipt, it as CampaignEvidenceContext } from "./statistical-heldout-Z9NROFFS.js";
|
|
4
4
|
import { H as SurfaceProposer, R as Scenario, S as JudgeConfig, a as CampaignResult, d as DispatchContext, j as MutableSurface, k as LabeledScenarioStore, p as Gate, v as GateResult } from "./types-BJz2CPTM.js";
|
|
5
|
-
import { $
|
|
5
|
+
import { $n as LoopProvenanceRecord, Ii as PremeasuredOptimizationBaseline, Li as RunOptimizationOptions, Pi as RunImprovementLoopResult, Zr as OptimizationMethod, aa as SearchHistoryReceipt, co as RunEvalOptions, do as CampaignCellRetryPolicy, ea as SearchHistoryAdmissionOptions, fo as RunCampaignOptions, ni as OptimizationMethodProvenance, ra as SearchHistoryCoverageRow, ri as OptimizationMethodResult, so as openAutoPr, ui as ComparisonCost, wo as CampaignStorage } from "./index-CtGf6e6X.js";
|
|
6
6
|
import { o as InsightReport } from "./insight-report-DETqPc_A.js";
|
|
7
7
|
import { n as HostedTenant } from "./client-vyYQg3bm.js";
|
|
8
8
|
//#region src/campaign/presets/run-final-comparison.d.ts
|
|
@@ -56,6 +56,11 @@ interface SelfImproveMethodProvenance {
|
|
|
56
56
|
totalDurationMs: number;
|
|
57
57
|
}
|
|
58
58
|
interface SelfImproveMethodResult<TScenario extends Scenario, TArtifact> extends Omit<SelfImproveProposerResult<TScenario, TArtifact>, 'mode' | 'baseline' | 'winner' | 'provenance' | 'generationsExplored' | 'cost' | 'raw' | 'optimization' | 'insight'> {
|
|
59
|
+
evidence?: {
|
|
60
|
+
baseline: EvidenceReceipt;
|
|
61
|
+
winner: EvidenceReceipt;
|
|
62
|
+
};
|
|
63
|
+
searchHistoryCoverage: SearchHistoryCoverageRow;
|
|
59
64
|
mode: 'method';
|
|
60
65
|
/** No final measurement exists when holdout is deferred. */
|
|
61
66
|
baseline: SelfImproveProposerResult<TScenario, TArtifact>['baseline'] | null;
|
|
@@ -149,7 +154,7 @@ type SelfImproveProgressEvent = {
|
|
|
149
154
|
mde: number;
|
|
150
155
|
underpowered: boolean;
|
|
151
156
|
};
|
|
152
|
-
interface SelfImproveOptions<TScenario extends Scenario, TArtifact> {
|
|
157
|
+
interface SelfImproveOptions<TScenario extends Scenario, TArtifact> extends SearchHistoryAdmissionOptions {
|
|
153
158
|
/**
|
|
154
159
|
* Your agent — a function that takes the current `MutableSurface`
|
|
155
160
|
* (typically a system prompt the loop is optimizing) plus the
|
|
@@ -320,6 +325,8 @@ interface SelfImproveOptions<TScenario extends Scenario, TArtifact> {
|
|
|
320
325
|
* return the bounded receipt on `searchHistory`. See
|
|
321
326
|
* `RunOptimizationOptions.searchLedger`. */
|
|
322
327
|
searchLedger?: RunOptimizationOptions<TScenario, TArtifact>['searchLedger'];
|
|
328
|
+
/** Complete-method final measurement receipts; authority and environment remain caller-owned. */
|
|
329
|
+
evidence?: CampaignEvidenceContext;
|
|
323
330
|
}
|
|
324
331
|
interface SelfImproveProposerResult<TScenario extends Scenario, TArtifact> {
|
|
325
332
|
mode: 'proposer';
|
|
@@ -495,4 +502,4 @@ interface DefinedAgentEval<TScenario extends Scenario, TArtifact> {
|
|
|
495
502
|
declare function defineAgentEval<TScenario extends Scenario, TArtifact>(defaults: DefineAgentEvalOptions<TScenario, TArtifact>): DefinedAgentEval<TScenario, TArtifact>;
|
|
496
503
|
//#endregion
|
|
497
504
|
export { SelfImproveMethodResult as _, DefinedAgentEval as a, SelfImproveMethodOptions as c, SelfImproveProposerOptions as d, SelfImproveProposerResult as f, SelfImproveMethodProvenance as g, selfImprove as h, DefineAgentEvalOptions as i, SelfImproveOptions as l, SelfImproveRunError as m, AgentEvalEvaluateOptions as n, defineAgentEval as o, SelfImproveResult as p, AgentEvalImproveOptions as r, SelfImproveBudget as s, AgentEvalAgent as t, SelfImproveProgressEvent as u };
|
|
498
|
-
//# sourceMappingURL=define-agent-eval-
|
|
505
|
+
//# sourceMappingURL=define-agent-eval-CqmlXUfQ.d.ts.map
|
|
@@ -0,0 +1 @@
|
|
|
1
|
+
{"version":3,"file":"define-agent-eval-CqmlXUfQ.d.ts","names":[],"sources":["../src/campaign/presets/run-final-comparison.ts","../src/contract/self-improve-method.ts","../src/contract/self-improve.ts","../src/contract/define-agent-eval.ts"],"mappings":";;;;;;;;UAMiB,uBAAuB,kBAAkB,UAAU,mBAC1D,KAAK,mBAAmB,WAAW;EAC3C,iBAAiB;EACjB,eAAe;EACf,sBACE,SAAS,gBACT,UAAU,WACV,KAAK,WAAW,mBAAmB,WAAW,+BAC3C,QAAQ;EACb,MAAM,KAAK,WAAW;EACtB;EACA;EACA,cAAc,QAAQ,gBAAgB,UAAU,mBAAmB;;;iBAI/C,mBAAmB,kBAAkB,UAAU,WACnE,MAAM,uBAAuB,WAAW,aAAU;;;;;;;;;;;UCWnC;EACf;EACA;EACA;EACA;EACA;EACA;EACA;EACA;EACA,oBAAoB,YAAY,0BAA0B;EAC1D;IACE;IACA;IACA;IACA;IACA;IACA;IACA;;EAEF,MAAM,QAAQ,kBAAkB;EAChC;EACA;EACA;EACA;EACA,MAAM;EACN;EACA;;UAGe,wBAAwB,kBAAkB,UAAU,mBAC3D,KACN,0BAA0B,WAAW;EAWvC;IAAa,UAAU;IAAiB,QAAQ;;EAChD,uBAAuB;EACvB;;EAEA,UAAU,0BAA0B,WAAW;EAC/C,QAAQ,KAAK,0BAA0B,WAAW;IAChD;;EAEF,YAAY;EACZ,cAAc,YAAY,0BAA0B,WAAW;;EAE/D,MAAM;;EAEN,YAAY;EACZ,UAAU,0BAA0B,WAAW;EAC/C,KAAK,QAAQ,kBAAkB,mBAAmB,WAAW;IAC3D;IACA,iBAAiB,0BAA0B,WAAW;IACtD,eAAe,0BAA0B,WAAW;IACpD;IACA,QAAQ;IACR,MAAM;IACN,WAAW,kBAAkB;;;;;UCxBhB;;;EAGf;;;;EAIA;;EAEA;;EAEA;;;EAGA;;;EAGA;;;;EAIA;;EAEA,mBAAmB;;;;;;;;;EASnB;;EAEA;;;;;EAKA;;KAGU;EACN;EAA0B;;EAC1B;EAA4B;EAAuB;;EACnD;EAA4B;EAAe;;EAC3C;EAA8B;EAAe;EAAuB;;EAGpE;EAAsB;EAAkB;;EACxC;EAAyB;EAAW;EAAY;EAAa;;UAElD,mBAAmB,kBAAkB,UAAU,mBACtD;;;;;;;;;;;;;;EAcR,QAAQ,SAAS,gBAAgB,UAAU,WAAW,KAAK,oBAAoB,QAAQ;;;;;;;EAQvF;;EAEA;;;EAIA,WAAW;;EAGX,OAAO,YAAY,WAAW;;;EAI9B,iBAAiB;;EAGjB,SAAS;;;;;;;;;;;;;EAcT,sBAAsB,gCAAgC,WAAW;;;;;EAMjE,WAAW,gBAAgB;;;;;;EAO3B,SAAS,mBAAmB,WAAW;;;EAIvC,qBAAqB;;;EAIrB,OAAO,KAAK,WAAW;;;;;;;;EASvB,cAAc,eAAe,gBAAgB,iBAAiB,mBAAmB;;;;EAKjF,UAAU;;;;EAKV;;;EAIA,gBAAgB,QAAQ,uBAAuB;;;;;EAM/C,iBAAiB;IACf,UAAU;IACV;IACA;;;;EAKF;;;;;;;EAQA,YAAY;;;EAIZ,cAAc,OAAO;;;;EAKrB;EACA;EACA;;;;;;;;;;;;;;EAeA,eAAe;;;EAIf,eAAe;;;;EAKf,eAAe;;EAGf;;;;;;;;EASA;;;;;;;;;EAUA,oBAAoB,uBAAuB,WAAW;;;EAItD,WAAW;;;;;;;EAQX,mBAAmB,uBAAuB,WAAW;;;;;;EAOrD,eAAe,uBAAuB,WAAW;;;;EAKjD,eAAe,uBAAuB,WAAW;;EAEjD,WAAW;;UAGI,0BAA0B,kBAAkB,UAAU;EACrE;;;;EAIA;IACE;IACA,aAAa;;;;;EAKf;IACE;IACA,aAAa;IACb,SAAS;;;IAGT;;;;IAIA;;;;;;EAMF;;;EAGA;;;;EAIA,YAAY;;EAEZ;;;EAGA;;EAEA;;EAEA;;EAEA,MAAM;;;EAGN,UAAU;;;EAGV,gBAAgB;;EAEhB;IACE;IACA,MAAM;IACN;IACA,aAAa;;;;;;;;EAQf,SAAS;;;;;EAKT,QAAQ;;;;;;EAMR,KAAK,yBAAyB,WAAW;;KAG/B,kBAAkB,kBAAkB,UAAU,aACtD,0BAA0B,WAAW,aACrC,wBAAwB,WAAW;KAE3B,yBAAyB,kBAAkB,UAAU,aAAa,KAC5E,mBAAmB,WAAW;EAG9B,QAAQ,mBAAmB,WAAW;EACtC;EACA,gBAAgB,QAAQ;;KAGd,2BAA2B,kBAAkB,UAAU,aAAa,KAC9E,mBAAmB,WAAW;EAG9B;EACA,gBAAgB,QAAQ;;;cAIb,4BAA4B;WAC9B,MAAM;WACN,UAAU;EAEnB,YAAY,gBAAgB,QAAQ;;;;;;;;;;;;;;;;;;;;;;;;;;;;iBAgMtB,YAAY,kBAAkB,UAAU,WACtD,MAAM,yBAAyB,WAAW,aACzC,QAAQ,wBAAwB,WAAW;iBAC9B,YAAY,kBAAkB,UAAU,WACtD,MAAM,2BAA2B,WAAW,aAC3C,QAAQ,0BAA0B,WAAW;iBAChC,YAAY,kBAAkB,UAAU,WACtD,MAAM,mBAAmB,WAAW,aACnC,QAAQ,kBAAkB,WAAW;;;KC1mB5B,eAAe,kBAAkB,UAAU,cACrD,SAAS,gBACT,UAAU,WACV,KAAK,oBACF,QAAQ;KAED,uBAAuB,kBAAkB,UAAU,aAAa,mBAC1E,WACA;UAGe,yBAAyB,kBAAkB,UAAU,mBAC5D,KACN,eAAe,WAAW;;EAI5B,YAAY;;EAEZ,UAAU;;EAEV,QAAQ,eAAe,WAAW;;EAElC,QAAQ,YAAY,WAAW;;EAE/B,SAAS,YAAY,WAAW;;EAEhC;;KAGU,wBAAwB,kBAAkB,UAAU,aAAa,KAC3E,QAAQ,mBAAmB,WAAW;EAGtC,SAAS,QAAQ;EACjB,eAAe,QAAQ;;UAGR,iBAAiB,kBAAkB,UAAU;;WAEnD,oBAAoB;;WAEpB,iBAAiB;;;;;EAK1B,SACE,OAAO,yBAAyB,WAAW,aAC1C,QAAQ,eAAe,WAAW;;;;;;;EAOrC,QACE,OAAO,wBAAwB,WAAW,aACzC,QAAQ,kBAAkB,WAAW;;;;;;;;;iBAU1B,gBAAgB,kBAAkB,UAAU,WAC1D,UAAU,uBAAuB,WAAW,aAC3C,iBAAiB,WAAW"}
|
|
@@ -1,5 +1,5 @@
|
|
|
1
1
|
import { a as resolveModelPricing } from "./metrics-Qv-cpptD.js";
|
|
2
|
-
import { _ as removeCredentialEnvironment, a as closeServer, c as waitForActiveHandlers, d as assertExternalOptimizerModelBudget, i as startExternalOptimizerModelProxy, o as listenLocal, r as runWithCleanup, s as sendJsonIfOpen, t as runExternalOptimizerProcess, v as resolveExternalOptimizerCallbackLimits, y as resolveExternalOptimizerProcessLimits } from "./external-optimizer-subprocess-
|
|
2
|
+
import { _ as removeCredentialEnvironment, a as closeServer, c as waitForActiveHandlers, d as assertExternalOptimizerModelBudget, i as startExternalOptimizerModelProxy, o as listenLocal, r as runWithCleanup, s as sendJsonIfOpen, t as runExternalOptimizerProcess, v as resolveExternalOptimizerCallbackLimits, y as resolveExternalOptimizerProcessLimits } from "./external-optimizer-subprocess-q3VzlGAO.js";
|
|
3
3
|
import { B as parseFindingSubject, P as coerceJson, j as RawAnalystFindingSchema } from "./kind-factory-BLvL-E44.js";
|
|
4
4
|
import { randomBytes } from "node:crypto";
|
|
5
5
|
import { createServer } from "node:http";
|
|
@@ -462,4 +462,4 @@ function isRecord(value) {
|
|
|
462
462
|
//#endregion
|
|
463
463
|
export { decodeRawFindingArray as n, createDspyRlmTraceEngine as t };
|
|
464
464
|
|
|
465
|
-
//# sourceMappingURL=dspy-rlm-engine-
|
|
465
|
+
//# sourceMappingURL=dspy-rlm-engine-DqjER2sV.js.map
|
|
@@ -1 +1 @@
|
|
|
1
|
-
{"version":3,"file":"dspy-rlm-engine-Caz2pl4L.js","names":["isRecord"],"sources":["../src/analyst/finding-codec.ts","../src/analyst/trace-tool-callback.ts","../src/analyst/dspy-rlm-engine.ts"],"sourcesContent":["/**\n * The raw-finding wire contract — one decoder for both sides of the process\n * boundary.\n *\n * A finding array crosses a process boundary twice: an optimizer bridge writes\n * it in Python, TypeScript reads it back. Two decoders written independently\n * accept different shapes, so Python could report a completed investigation\n * whose rows TypeScript then dropped — the paid work disappearing between two\n * \"successes\". This module is the TypeScript half; `finding_contract.json`\n * carries the same rules to the Python half, generated from the schema here.\n *\n * Decoding never invents an empty result. A value that is not an array of rows\n * fails with the exact type it was, and a malformed row is reported with its\n * index and path while its valid siblings survive.\n */\n\nimport { type RawAnalystFinding, RawAnalystFindingSchema } from './finding-signature'\nimport { parseFindingSubject } from './finding-subject'\nimport { coerceJson } from './parse-tolerant'\n\n/**\n * Version of the wire contract both languages implement. Bump when the\n * accepted row shape changes; the Python package pins the same value, and the\n * emitted contract carries it so a mismatched pair is visible.\n */\nexport const FINDING_WIRE_CONTRACT_VERSION = 1\n\n/** Why one row was refused. Stable across languages: Python reports the same codes. */\nexport type FindingRejectionCode = 'not-an-object' | 'schema' | 'invalid-subject' | 'row-limit'\n\n/** A refused row, named precisely enough to repair or report. */\nexport interface RejectedFindingRow {\n /** Position in the submitted array. */\n readonly index: number\n /** Dotted path to the offending field, `''` for the row itself. */\n readonly path: string\n readonly code: FindingRejectionCode\n readonly message: string\n}\n\nexport interface DecodedFindingArray {\n readonly accepted: RawAnalystFinding[]\n readonly rejected: RejectedFindingRow[]\n /**\n * Set when the value was not a finding array at all — the array itself is\n * the failure, so `accepted` and `rejected` are both empty and a caller must\n * not read the result as \"no findings\".\n */\n readonly topLevelError?: string\n}\n\n/**\n * Rows past this count are refused rather than validated. A model that emits\n * thousands of rows has lost the plot, and the diagnostics for them would\n * dwarf the answer they came with.\n */\nexport const MAX_FINDING_ROWS = 500\n\n/**\n * Decode a submitted findings value into accepted rows plus per-row\n * diagnostics.\n *\n * Accepts an array of rows, or a string carrying one (a JSON array, possibly\n * fenced). Anything else — an object, a number, null, undefined — sets\n * `topLevelError` naming the type received: a caller that asked for findings\n * and got a number has a defect to report, not an empty result to record.\n */\nexport function decodeRawFindingArray(value: unknown): DecodedFindingArray {\n const rows = findingRows(value)\n if (typeof rows === 'string') return { accepted: [], rejected: [], topLevelError: rows }\n\n const accepted: RawAnalystFinding[] = []\n const rejected: RejectedFindingRow[] = []\n for (const [index, row] of rows.entries()) {\n if (index >= MAX_FINDING_ROWS) {\n rejected.push({\n index,\n path: '',\n code: 'row-limit',\n message: `findings array exceeds ${MAX_FINDING_ROWS} rows`,\n })\n continue\n }\n const decoded = decodeRow(row, index)\n if ('finding' in decoded) accepted.push(decoded.finding)\n else rejected.push(decoded.rejection)\n }\n return { accepted, rejected }\n}\n\nfunction decodeRow(\n row: unknown,\n index: number,\n): { finding: RawAnalystFinding } | { rejection: RejectedFindingRow } {\n if (row === null || typeof row !== 'object' || Array.isArray(row)) {\n return {\n rejection: {\n index,\n path: '',\n code: 'not-an-object',\n message: `finding row must be an object, received ${describe(row)}`,\n },\n }\n }\n const parsed = RawAnalystFindingSchema.safeParse(row)\n if (!parsed.success) {\n const issue = parsed.error.issues[0]\n const path = (issue?.path ?? []).join('.')\n // The schema refines `subject` against the grammar; name that refusal\n // separately so a repair turn can be told which rule the row broke.\n const code: FindingRejectionCode =\n path === 'subject' && parseFindingSubject((row as { subject?: string }).subject) === null\n ? 'invalid-subject'\n : 'schema'\n return {\n rejection: { index, path, code, message: issue?.message ?? 'row does not match the schema' },\n }\n }\n return { finding: parsed.data }\n}\n\n/**\n * Narrow a submitted value to candidate rows, or return the message naming why\n * it is not a findings array at all.\n */\nfunction findingRows(value: unknown): unknown[] | string {\n if (Array.isArray(value)) return value\n if (typeof value === 'string') {\n const parsed = coerceJson(value)\n if (parsed === undefined) {\n return 'findings must be a JSON array; the string received is not JSON'\n }\n if (Array.isArray(parsed)) return parsed\n const unwrapped = findingsProperty(parsed)\n if (unwrapped !== undefined) return unwrapped\n return `findings must be a JSON array, received ${describe(parsed)}`\n }\n const unwrapped = findingsProperty(value)\n if (unwrapped !== undefined) return unwrapped\n return `findings must be an array, received ${describe(value)}`\n}\n\n/** `{ findings: [...] }` is the one wrapper models emit often enough to unwrap. */\nfunction findingsProperty(value: unknown): unknown[] | undefined {\n if (value === null || typeof value !== 'object' || Array.isArray(value)) return undefined\n const inner = (value as Record<string, unknown>).findings\n return Array.isArray(inner) ? inner : undefined\n}\n\nfunction describe(value: unknown): string {\n if (value === null) return 'null'\n if (Array.isArray(value)) return 'an array'\n return `a ${typeof value}`\n}\n\n/** One line per refused row, for a repair prompt or an error message. */\nexport function describeRejectedRows(rejected: readonly RejectedFindingRow[]): string {\n return rejected\n .map((row) => `row ${row.index}${row.path ? ` field '${row.path}'` : ''}: ${row.message}`)\n .join('\\n')\n}\n","import { randomBytes } from 'node:crypto'\nimport { createServer, type IncomingMessage, type ServerResponse } from 'node:http'\nimport {\n type ExternalOptimizerCallbackLimits,\n resolveExternalOptimizerCallbackLimits,\n} from '../campaign/external-optimizer-contracts'\nimport {\n closeServer,\n listenLocal,\n sendJsonIfOpen,\n waitForActiveHandlers,\n} from '../campaign/external-optimizer-http'\nimport type { TraceAnalysisToolDescriptor } from '../trace-analyst/tools'\n\nexport interface TraceToolCallback {\n url: string\n token: string\n calls: () => number\n close: () => Promise<void>\n}\n\nexport type TraceToolCallbackLimits = ExternalOptimizerCallbackLimits\n\n/** Expose one bounded trace-tool set only on an authenticated loopback socket. */\nexport async function startTraceToolCallback(args: {\n tools: readonly TraceAnalysisToolDescriptor[]\n maxCalls: number\n /** Trace-tool request/response byte limits. Omitted fields use finite defaults. */\n limits?: Partial<TraceToolCallbackLimits>\n signal?: AbortSignal\n}): Promise<TraceToolCallback> {\n if (!Number.isSafeInteger(args.maxCalls) || args.maxCalls <= 0) {\n throw new TypeError('trace tool callback maxCalls must be a positive safe integer')\n }\n const limits = resolveExternalOptimizerCallbackLimits(args.limits, 'trace tool callback limits')\n args.signal?.throwIfAborted()\n const byName = new Map(args.tools.map((tool) => [tool.name, tool]))\n if (byName.size !== args.tools.length) {\n throw new Error('trace tool callback received duplicate tool names')\n }\n\n const token = randomBytes(32).toString('hex')\n let calls = 0\n let accepting = true\n let closePromise: Promise<void> | undefined\n const activeControllers = new Set<AbortController>()\n const activeHandlers = new Set<Promise<void>>()\n const server = createServer((request, response) => {\n if (!accepting) {\n sendJsonIfOpen(response, 503, { error: 'trace tool callback is closing' })\n return\n }\n const controller = new AbortController()\n const abortRequest = (): void => {\n request.destroy()\n response.destroy()\n }\n activeControllers.add(controller)\n controller.signal.addEventListener('abort', abortRequest, { once: true })\n\n let handler!: Promise<void>\n handler = handleRequest(request, response, controller.signal).finally(() => {\n controller.signal.removeEventListener('abort', abortRequest)\n activeControllers.delete(controller)\n activeHandlers.delete(handler)\n })\n activeHandlers.add(handler)\n void handler.catch(() => undefined)\n })\n const port = await listenLocal(server)\n const close = (): Promise<void> => {\n closePromise ??= closeCallback()\n return closePromise\n }\n const onAbort = (): void => {\n void close().catch(() => undefined)\n }\n args.signal?.addEventListener('abort', onAbort, { once: true })\n if (args.signal?.aborted) onAbort()\n\n return {\n url: `http://127.0.0.1:${port}/call`,\n token,\n calls: () => calls,\n close,\n }\n\n async function handleRequest(\n request: IncomingMessage,\n response: ServerResponse,\n signal: AbortSignal,\n ): Promise<void> {\n try {\n if (request.method !== 'POST' || request.url !== '/call') {\n sendJsonIfOpen(response, 404, { error: 'not found' })\n return\n }\n if (request.headers.authorization !== `Bearer ${token}`) {\n sendJsonIfOpen(response, 401, { error: 'unauthorized' })\n return\n }\n if (calls >= args.maxCalls) {\n sendJsonIfOpen(response, 429, { error: 'trace tool call limit reached' })\n return\n }\n const body = await readJson(request, limits.maxRequestBytes)\n if (!isRecord(body) || typeof body.name !== 'string' || !('args' in body)) {\n sendJsonIfOpen(response, 400, { error: 'name and args are required' })\n return\n }\n const tool = byName.get(body.name)\n if (!tool) {\n sendJsonIfOpen(response, 404, { error: `unknown trace tool '${body.name}'` })\n return\n }\n calls += 1\n const result = await tool.handler(body.args, { signal })\n const encoded = JSON.stringify({ result })\n if (Buffer.byteLength(encoded) > limits.maxResponseBytes) {\n sendJsonIfOpen(response, 413, { error: 'trace tool response too large' })\n return\n }\n response.writeHead(200, {\n 'content-type': 'application/json; charset=utf-8',\n 'content-length': String(Buffer.byteLength(encoded)),\n })\n response.end(encoded)\n } catch (error) {\n sendJsonIfOpen(response, signal.aborted ? 499 : 400, {\n error: error instanceof Error ? error.message : String(error),\n })\n }\n }\n\n async function closeCallback(): Promise<void> {\n args.signal?.removeEventListener('abort', onAbort)\n accepting = false\n const closingServer = closeServer(server)\n server.closeIdleConnections?.()\n for (const controller of activeControllers) controller.abort()\n const [serverResult] = await Promise.allSettled([\n closingServer,\n waitForActiveHandlers(activeHandlers),\n ])\n if (activeControllers.size !== 0 || activeHandlers.size !== 0) {\n throw new Error('trace tool callback closed with active requests')\n }\n if (serverResult?.status === 'rejected') throw serverResult.reason\n }\n}\n\nfunction readJson(request: IncomingMessage, maxRequestBytes: number): Promise<unknown> {\n return new Promise((resolve, reject) => {\n let size = 0\n const chunks: Buffer[] = []\n request.on('data', (chunk: Buffer) => {\n size += chunk.length\n if (size > maxRequestBytes) {\n reject(new Error('trace tool request too large'))\n request.destroy()\n return\n }\n chunks.push(chunk)\n })\n request.on('error', reject)\n request.on('end', () => {\n try {\n resolve(JSON.parse(Buffer.concat(chunks).toString('utf8')))\n } catch (error) {\n reject(error)\n }\n })\n })\n}\n\nfunction isRecord(value: unknown): value is Record<string, unknown> {\n return typeof value === 'object' && value !== null && !Array.isArray(value)\n}\n","import {\n assertExternalOptimizerModelBudget,\n type ExternalOptimizerModelCall,\n type ExternalOptimizerModelExecutionObservation,\n type ExternalOptimizerModelProxy,\n type ExternalOptimizerRunnerCommand,\n removeCredentialEnvironment,\n resolveExternalOptimizerCallbackLimits,\n resolveExternalOptimizerProcessLimits,\n} from '../campaign/external-optimizer-contracts'\nimport { startExternalOptimizerModelProxy } from '../campaign/external-optimizer-model-proxy'\nimport { runWithCleanup } from '../campaign/external-optimizer-resources'\nimport { runExternalOptimizerProcess } from '../campaign/external-optimizer-subprocess'\nimport type { CustomTokenPricing } from '../cost-ledger'\nimport { resolveModelPricing } from '../metrics'\nimport {\n DEFAULT_TRACE_ANALYST_OUTPUT_TOKENS,\n type TraceAnalysisEngine,\n type TraceAnalysisEngineResult,\n} from './engine'\nimport { decodeRawFindingArray } from './finding-codec'\nimport { startTraceToolCallback, type TraceToolCallbackLimits } from './trace-tool-callback'\n\nconst DEFAULT_TIMEOUT_MS = 10 * 60_000\nconst DEFAULT_MAX_COST_USD = 1\nconst DEFAULT_MAX_MODEL_REQUEST_BYTES = 16 * 1024 * 1024\nconst DEFAULT_MAX_MODEL_RESPONSE_BYTES = 4 * 1024 * 1024\nconst DEFAULT_TRACE_TOOL_TIMEOUT_MS = 60_000\nconst MAX_TIMER_DELAY_MS = 2_147_483_647\nconst BRIDGE_MODULE = 'agent_eval_rpc.dspy_rlm_bridge'\n/** Bumped whenever this engine's execution behavior changes. */\nconst DSPY_RLM_ENGINE_VERSION = '1.0.0'\n\nexport interface DspyRlmTraceEngineOptions {\n /** Caller-owned execution path. Agent Eval never receives provider credentials. */\n call: ExternalOptimizerModelCall\n /** Stable public identity for the caller-owned path, such as an AgentProfile digest. */\n callRef: string\n /** Persist the finite execution record returned for every admitted call. */\n recordExecution: (observation: ExternalOptimizerModelExecutionObservation) => void\n model: string\n /** Exact provider rates. Required when the model is absent from the pricing table. */\n pricing?: CustomTokenPricing\n /** Maximum provider spend for one investigation. Default: 1 USD. */\n maxCostUsd?: number\n /** Controller response cap. Default: {@link DEFAULT_TRACE_ANALYST_OUTPUT_TOKENS}. */\n maxOutputTokens?: number\n /**\n * Thinking tokens one controller turn may bill on top of its completion.\n * A reasoning model bills these beyond `maxOutputTokens`, so the cost\n * reservation must cover them. Default: four times the completion cap.\n */\n maxReasoningTokens?: number\n /** Maximum caller-owned model invocations. Default derives from the analysis limits. */\n maxModelRequests?: number\n /** Maximum model request bytes. Default: 16 MiB. */\n maxModelRequestBytes?: number\n /** Maximum model response bytes. Default: 4 MiB. */\n maxModelResponseBytes?: number\n /** Deadline for one caller-owned model invocation. Default: the whole analysis deadline. */\n modelRequestTimeoutMs?: number\n /** Trace-tool loopback request/response byte limits. */\n traceToolLimits?: Partial<TraceToolCallbackLimits>\n /** Deadline for one Python-to-Node trace-tool call. Default: 60 seconds. */\n traceToolTimeoutMs?: number\n /**\n * How the controller's reasoning and code fields are obtained.\n *\n * `tolerant` parses marker output strictly first, then recovers the fields\n * deterministically from prose plus a fenced code block — the shape coding\n * models naturally emit — at no extra model cost. `two-step` extracts with a\n * second call per turn. `chat` accepts marker output only. Default:\n * `tolerant`.\n */\n controlAdapter?: 'chat' | 'two-step' | 'tolerant'\n /** Python command used to load agent-eval-rpc[dspy]. Default: python. */\n runner?: ExternalOptimizerRunnerCommand\n /** Whole investigation deadline. Default: 10 minutes. */\n timeoutMs?: number\n}\n\n/** Use the official DSPy RLM as a bounded recursive trace-analysis engine. */\nexport function createDspyRlmTraceEngine(options: DspyRlmTraceEngineOptions): TraceAnalysisEngine {\n assertOptions(options)\n const maxOutputTokens = options.maxOutputTokens ?? DEFAULT_TRACE_ANALYST_OUTPUT_TOKENS\n if (!Number.isSafeInteger(maxOutputTokens) || maxOutputTokens <= 0) {\n throw new TypeError('DSPy RLM maxOutputTokens must be a positive safe integer')\n }\n const controlAdapter = options.controlAdapter ?? 'tolerant'\n const maxReasoningTokens = options.maxReasoningTokens ?? maxOutputTokens * 4\n if (!Number.isSafeInteger(maxReasoningTokens) || maxReasoningTokens < 0) {\n throw new TypeError('DSPy RLM maxReasoningTokens must be a non-negative safe integer')\n }\n const timeoutMs = options.timeoutMs ?? DEFAULT_TIMEOUT_MS\n assertTimerDelay(timeoutMs, 'timeoutMs')\n const maxCostUsd = options.maxCostUsd ?? DEFAULT_MAX_COST_USD\n if (!Number.isFinite(maxCostUsd) || maxCostUsd <= 0) {\n throw new TypeError('DSPy RLM maxCostUsd must be positive and finite')\n }\n const pricing = options.pricing ?? pricingForModel(options.model)\n const maxModelRequestBytes = options.maxModelRequestBytes ?? DEFAULT_MAX_MODEL_REQUEST_BYTES\n const maxModelResponseBytes = options.maxModelResponseBytes ?? DEFAULT_MAX_MODEL_RESPONSE_BYTES\n const modelRequestTimeoutMs = options.modelRequestTimeoutMs ?? timeoutMs\n const traceToolTimeoutMs = options.traceToolTimeoutMs ?? DEFAULT_TRACE_TOOL_TIMEOUT_MS\n assertExternalOptimizerModelBudget(\n {\n maxCostUsd,\n maxRequests: options.maxModelRequests ?? 1,\n maxRequestBytes: maxModelRequestBytes,\n maxResponseBytes: maxModelResponseBytes,\n maxOutputTokensPerRequest: maxOutputTokens,\n maxReasoningTokensPerRequest: maxReasoningTokens,\n pricing,\n requestTimeoutMs: modelRequestTimeoutMs,\n },\n 'DSPy RLM model limits',\n )\n resolveExternalOptimizerCallbackLimits(options.traceToolLimits, 'DSPy RLM trace tool limits')\n assertTimerDelay(traceToolTimeoutMs, 'traceToolTimeoutMs')\n\n const runner = sanitizedRunner(options.runner)\n const processLimits = resolveExternalOptimizerProcessLimits(runner?.limits)\n return {\n id: 'dspy-rlm',\n description: 'Official DSPy RLM with bounded trace tools and metered model calls.',\n model: options.model,\n version: DSPY_RLM_ENGINE_VERSION,\n executionConfig: {\n bridge_module: BRIDGE_MODULE,\n call_ref: options.callRef,\n model: options.model,\n pricing: { ...pricing },\n max_cost_usd: maxCostUsd,\n max_output_tokens: maxOutputTokens,\n max_reasoning_tokens: maxReasoningTokens,\n control_adapter: controlAdapter,\n timeout_ms: timeoutMs,\n max_model_requests: options.maxModelRequests ?? null,\n max_request_bytes: maxModelRequestBytes,\n max_response_bytes: maxModelResponseBytes,\n model_request_timeout_ms: modelRequestTimeoutMs,\n trace_tool_limits: options.traceToolLimits ?? null,\n trace_tool_timeout_ms: traceToolTimeoutMs,\n process_limits: processLimits,\n runner: runner ? 'caller-supplied' : 'default',\n runner_command: runner?.command ?? null,\n },\n async analyze(request) {\n const callback = await startTraceToolCallback({\n tools: request.tools,\n maxCalls: request.limits.maxToolCalls,\n ...(options.traceToolLimits ? { limits: options.traceToolLimits } : {}),\n ...(request.signal ? { signal: request.signal } : {}),\n })\n let modelProxy: ExternalOptimizerModelProxy | undefined\n const modelExecutions: ExternalOptimizerModelExecutionObservation[] = []\n const result = await runWithCleanup({\n label: 'DSPy RLM trace-analysis resources',\n run: async () => {\n modelProxy = await startExternalOptimizerModelProxy({\n call: options.call,\n callRef: options.callRef,\n recordExecution: (observation) => {\n modelExecutions.push(structuredClone(observation))\n options.recordExecution(observation)\n },\n model: options.model,\n budget: {\n maxCostUsd,\n maxRequests: resolveMaxModelRequests(options.maxModelRequests, request.limits),\n maxRequestBytes: maxModelRequestBytes,\n maxResponseBytes: maxModelResponseBytes,\n maxOutputTokensPerRequest: maxOutputTokens,\n maxReasoningTokensPerRequest: maxReasoningTokens,\n requestTimeoutMs: modelRequestTimeoutMs,\n pricing,\n },\n costLedger: request.costLedger,\n channel: 'analyst',\n phase: request.costPhase,\n actor: request.analystId,\n ...(request.costTags ? { tags: request.costTags } : {}),\n ...(request.signal ? { signal: request.signal } : {}),\n })\n request.log?.('trace analyst engine started', {\n engine: 'dspy-rlm',\n model: options.model,\n tools: request.tools.map((tool) => tool.name),\n limits: request.limits,\n })\n const raw = await runExternalOptimizerProcess<unknown>({\n label: 'DSPy RLM trace analysis',\n tempPrefix: 'agent-eval-dspy-rlm-',\n module: BRIDGE_MODULE,\n input: {\n operation: 'analyze',\n question: request.question,\n instructions: request.instructions,\n modelProxy: {\n baseUrl: modelProxy.baseUrl,\n apiKey: modelProxy.apiKey,\n model: options.model,\n maxOutputTokens,\n },\n toolCallback: {\n url: callback.url,\n token: callback.token,\n timeoutMs: traceToolTimeoutMs,\n },\n toolSpecs: request.tools.map(({ name, description, parameters }) => ({\n name,\n description,\n parameters,\n })),\n controlAdapter,\n limits: {\n maxIterations: request.limits.maxIterations,\n maxLlmCalls: request.limits.maxLlmCalls,\n maxOutputChars: request.limits.maxOutputChars,\n },\n // Omitted entirely when the caller supplied none, so a request\n // without structured inputs sends the payload it always sent.\n ...(request.taskInputs ? { taskInputs: request.taskInputs } : {}),\n },\n ...(runner ? { runner } : {}),\n additionalArgs: ['--max-input-bytes', String(processLimits.maxInputBytes)],\n timeoutMs,\n ...(request.signal ? { signal: request.signal } : {}),\n })\n const parsed = parseBridgeOutput(raw, (index, reason) => {\n request.log?.('finding rejected: bridge row failed schema validation', {\n engine: 'dspy-rlm',\n index,\n reason,\n })\n })\n const successfulCompletions = modelProxy.successfulCompletions()\n const requestAttempts = modelProxy.requestAttempts()\n if (parsed.modelCalls !== successfulCompletions) {\n throw new Error(\n `DSPy RLM reported ${parsed.modelCalls} model calls, but the provider proxy recorded ${successfulCompletions}`,\n )\n }\n modelProxy.assertExecutionComplete()\n return {\n ...parsed,\n toolCalls: callback.calls(),\n runtime: {\n ...parsed.runtime,\n modelRequestAttempts: requestAttempts,\n modelSuccessfulCompletions: successfulCompletions,\n modelExecutions,\n },\n } satisfies TraceAnalysisEngineResult\n },\n cleanup: async () => {\n const results = await Promise.allSettled([\n ...(modelProxy ? [modelProxy.close()] : []),\n callback.close(),\n ])\n const errors = results.flatMap((entry) =>\n entry.status === 'rejected' ? [entry.reason] : [],\n )\n if (errors.length > 0) {\n throw new AggregateError(errors, 'DSPy RLM resource cleanup failed')\n }\n },\n })\n request.log?.('trace analyst engine completed', {\n engine: 'dspy-rlm',\n model_calls: result.modelCalls,\n model_request_attempts: result.runtime.modelRequestAttempts,\n tool_calls: result.toolCalls,\n findings: result.findings.length,\n })\n return result\n },\n }\n}\n\nfunction parseBridgeOutput(\n value: unknown,\n onRejectedFinding: (index: number, reason: string) => void,\n): Omit<TraceAnalysisEngineResult, 'toolCalls'> {\n if (!isRecord(value)) throw new Error('DSPy RLM bridge output must be an object')\n if (typeof value.answer !== 'string' || !value.answer.trim()) {\n throw new Error('DSPy RLM bridge returned no answer')\n }\n // Findings are model output: one malformed row is model noise, not a bridge\n // fault, and the rest of the paid investigation must survive it. The codec\n // is the same decoder the Python side runs, so a row this bridge accepts is\n // a row the optimizer could report, and vice versa.\n const decoded = decodeRawFindingArray(value.findings)\n if (decoded.topLevelError !== undefined) {\n throw new Error(`DSPy RLM bridge findings must be an array: ${decoded.topLevelError}`)\n }\n const findings = decoded.accepted\n const rejectedFindings = decoded.rejected.length\n for (const rejection of decoded.rejected) {\n onRejectedFinding(\n rejection.index,\n `${rejection.code}${rejection.path ? ` at ${rejection.path}` : ''}: ${rejection.message}`,\n )\n }\n if (!Array.isArray(value.trajectory)) {\n throw new Error('DSPy RLM bridge trajectory must be an array')\n }\n if (!Number.isSafeInteger(value.modelCalls) || (value.modelCalls as number) <= 0) {\n throw new Error('DSPy RLM bridge modelCalls must be a positive safe integer')\n }\n if (!isRecord(value.runtime)) {\n throw new Error('DSPy RLM bridge runtime must be an object')\n }\n return {\n answer: value.answer,\n findings,\n trajectory: value.trajectory,\n modelCalls: value.modelCalls as number,\n runtime: { ...value.runtime, rejectedFindings },\n }\n}\n\nfunction assertOptions(options: DspyRlmTraceEngineOptions): void {\n for (const [name, value] of [\n ['callRef', options.callRef],\n ['model', options.model],\n ] as const) {\n if (typeof value !== 'string' || !value.trim() || value !== value.trim()) {\n throw new TypeError(`DSPy RLM ${name} must be a trimmed non-empty string`)\n }\n }\n if (typeof options.call !== 'function') {\n throw new TypeError('DSPy RLM call must be a function')\n }\n if (typeof options.recordExecution !== 'function') {\n throw new TypeError('DSPy RLM recordExecution must be a function')\n }\n if (\n options.maxModelRequests !== undefined &&\n (!Number.isSafeInteger(options.maxModelRequests) || options.maxModelRequests <= 0)\n ) {\n throw new TypeError('DSPy RLM maxModelRequests must be a positive safe integer')\n }\n}\n\nfunction resolveMaxModelRequests(\n configured: number | undefined,\n limits: { maxIterations: number; maxLlmCalls: number },\n): number {\n if (configured !== undefined) return configured\n const derived = limits.maxIterations + limits.maxLlmCalls + 1\n if (!Number.isSafeInteger(derived) || derived <= 0) {\n throw new Error('DSPy RLM analysis limits produce an invalid model request limit')\n }\n return derived\n}\n\nfunction assertTimerDelay(value: number, field: string): void {\n if (!Number.isSafeInteger(value) || value <= 0 || value > MAX_TIMER_DELAY_MS) {\n throw new TypeError(`DSPy RLM ${field} must be between 1 and ${MAX_TIMER_DELAY_MS}`)\n }\n}\n\nfunction pricingForModel(model: string): CustomTokenPricing {\n const pricing = resolveModelPricing(model)\n if (!pricing) {\n throw new Error(\n `no pricing is configured for '${model}'; provide DspyRlmTraceEngineOptions.pricing`,\n )\n }\n return {\n inputUsdPerMillion: pricing.input * 1_000,\n outputUsdPerMillion: pricing.output * 1_000,\n }\n}\n\nfunction sanitizedRunner(\n runner: ExternalOptimizerRunnerCommand | undefined,\n): ExternalOptimizerRunnerCommand | undefined {\n if (!runner) return undefined\n return {\n ...(runner.command ? { command: runner.command } : {}),\n ...(runner.args ? { args: runner.args } : {}),\n ...(runner.env ? { env: removeCredentialEnvironment(runner.env) } : {}),\n ...(runner.limits ? { limits: { ...runner.limits } } : {}),\n }\n}\n\nfunction isRecord(value: unknown): value is Record<string, unknown> {\n return typeof value === 'object' && value !== null && !Array.isArray(value)\n}\n"],"mappings":";;;;;;;;;;;;;;AAmEA,SAAgB,sBAAsB,OAAqC;CACzE,MAAM,OAAO,YAAY,KAAK;CAC9B,IAAI,OAAO,SAAS,UAAU,OAAO;EAAE,UAAU,CAAC;EAAG,UAAU,CAAC;EAAG,eAAe;CAAK;CAEvF,MAAM,WAAgC,CAAC;CACvC,MAAM,WAAiC,CAAC;CACxC,KAAK,MAAM,CAAC,OAAO,QAAQ,KAAK,QAAQ,GAAG;EACzC,IAAI,SAAA,KAA2B;GAC7B,SAAS,KAAK;IACZ;IACA,MAAM;IACN,MAAM;IACN,SAAS;GACX,CAAC;GACD;EACF;EACA,MAAM,UAAU,UAAU,KAAK,KAAK;EACpC,IAAI,aAAa,SAAS,SAAS,KAAK,QAAQ,OAAO;OAClD,SAAS,KAAK,QAAQ,SAAS;CACtC;CACA,OAAO;EAAE;EAAU;CAAS;AAC9B;AAEA,SAAS,UACP,KACA,OACoE;CACpE,IAAI,QAAQ,QAAQ,OAAO,QAAQ,YAAY,MAAM,QAAQ,GAAG,GAC9D,OAAO,EACL,WAAW;EACT;EACA,MAAM;EACN,MAAM;EACN,SAAS,2CAA2C,SAAS,GAAG;CAClE,EACF;CAEF,MAAM,SAAS,wBAAwB,UAAU,GAAG;CACpD,IAAI,CAAC,OAAO,SAAS;EACnB,MAAM,QAAQ,OAAO,MAAM,OAAO;EAClC,MAAM,QAAQ,OAAO,QAAQ,CAAC,EAAA,CAAG,KAAK,GAAG;EAOzC,OAAO,EACL,WAAW;GAAE;GAAO;GAAM,MAJ1B,SAAS,aAAa,oBAAqB,IAA6B,OAAO,MAAM,OACjF,oBACA;GAE4B,SAAS,OAAO,WAAW;EAAgC,EAC7F;CACF;CACA,OAAO,EAAE,SAAS,OAAO,KAAK;AAChC;;;;;AAMA,SAAS,YAAY,OAAoC;CACvD,IAAI,MAAM,QAAQ,KAAK,GAAG,OAAO;CACjC,IAAI,OAAO,UAAU,UAAU;EAC7B,MAAM,SAAS,WAAW,KAAK;EAC/B,IAAI,WAAW,KAAA,GACb,OAAO;EAET,IAAI,MAAM,QAAQ,MAAM,GAAG,OAAO;EAClC,MAAM,YAAY,iBAAiB,MAAM;EACzC,IAAI,cAAc,KAAA,GAAW,OAAO;EACpC,OAAO,2CAA2C,SAAS,MAAM;CACnE;CACA,MAAM,YAAY,iBAAiB,KAAK;CACxC,IAAI,cAAc,KAAA,GAAW,OAAO;CACpC,OAAO,uCAAuC,SAAS,KAAK;AAC9D;;AAGA,SAAS,iBAAiB,OAAuC;CAC/D,IAAI,UAAU,QAAQ,OAAO,UAAU,YAAY,MAAM,QAAQ,KAAK,GAAG,OAAO,KAAA;CAChF,MAAM,QAAS,MAAkC;CACjD,OAAO,MAAM,QAAQ,KAAK,IAAI,QAAQ,KAAA;AACxC;AAEA,SAAS,SAAS,OAAwB;CACxC,IAAI,UAAU,MAAM,OAAO;CAC3B,IAAI,MAAM,QAAQ,KAAK,GAAG,OAAO;CACjC,OAAO,KAAK,OAAO;AACrB;;;;ACjIA,eAAsB,uBAAuB,MAMd;CAC7B,IAAI,CAAC,OAAO,cAAc,KAAK,QAAQ,KAAK,KAAK,YAAY,GAC3D,MAAM,IAAI,UAAU,8DAA8D;CAEpF,MAAM,SAAS,uCAAuC,KAAK,QAAQ,4BAA4B;CAC/F,KAAK,QAAQ,eAAe;CAC5B,MAAM,SAAS,IAAI,IAAI,KAAK,MAAM,KAAK,SAAS,CAAC,KAAK,MAAM,IAAI,CAAC,CAAC;CAClE,IAAI,OAAO,SAAS,KAAK,MAAM,QAC7B,MAAM,IAAI,MAAM,mDAAmD;CAGrE,MAAM,QAAQ,YAAY,EAAE,CAAC,CAAC,SAAS,KAAK;CAC5C,IAAI,QAAQ;CACZ,IAAI,YAAY;CAChB,IAAI;CACJ,MAAM,oCAAoB,IAAI,IAAqB;CACnD,MAAM,iCAAiB,IAAI,IAAmB;CAC9C,MAAM,SAAS,cAAc,SAAS,aAAa;EACjD,IAAI,CAAC,WAAW;GACd,eAAe,UAAU,KAAK,EAAE,OAAO,iCAAiC,CAAC;GACzE;EACF;EACA,MAAM,aAAa,IAAI,gBAAgB;EACvC,MAAM,qBAA2B;GAC/B,QAAQ,QAAQ;GAChB,SAAS,QAAQ;EACnB;EACA,kBAAkB,IAAI,UAAU;EAChC,WAAW,OAAO,iBAAiB,SAAS,cAAc,EAAE,MAAM,KAAK,CAAC;EAExE,IAAI;EACJ,UAAU,cAAc,SAAS,UAAU,WAAW,MAAM,CAAC,CAAC,cAAc;GAC1E,WAAW,OAAO,oBAAoB,SAAS,YAAY;GAC3D,kBAAkB,OAAO,UAAU;GACnC,eAAe,OAAO,OAAO;EAC/B,CAAC;EACD,eAAe,IAAI,OAAO;EAC1B,QAAa,YAAY,KAAA,CAAS;CACpC,CAAC;CACD,MAAM,OAAO,MAAM,YAAY,MAAM;CACrC,MAAM,cAA6B;EACjC,iBAAiB,cAAc;EAC/B,OAAO;CACT;CACA,MAAM,gBAAsB;EAC1B,MAAW,CAAC,CAAC,YAAY,KAAA,CAAS;CACpC;CACA,KAAK,QAAQ,iBAAiB,SAAS,SAAS,EAAE,MAAM,KAAK,CAAC;CAC9D,IAAI,KAAK,QAAQ,SAAS,QAAQ;CAElC,OAAO;EACL,KAAK,oBAAoB,KAAK;EAC9B;EACA,aAAa;EACb;CACF;CAEA,eAAe,cACb,SACA,UACA,QACe;EACf,IAAI;GACF,IAAI,QAAQ,WAAW,UAAU,QAAQ,QAAQ,SAAS;IACxD,eAAe,UAAU,KAAK,EAAE,OAAO,YAAY,CAAC;IACpD;GACF;GACA,IAAI,QAAQ,QAAQ,kBAAkB,UAAU,SAAS;IACvD,eAAe,UAAU,KAAK,EAAE,OAAO,eAAe,CAAC;IACvD;GACF;GACA,IAAI,SAAS,KAAK,UAAU;IAC1B,eAAe,UAAU,KAAK,EAAE,OAAO,gCAAgC,CAAC;IACxE;GACF;GACA,MAAM,OAAO,MAAM,SAAS,SAAS,OAAO,eAAe;GAC3D,IAAI,CAACA,WAAS,IAAI,KAAK,OAAO,KAAK,SAAS,YAAY,EAAE,UAAU,OAAO;IACzE,eAAe,UAAU,KAAK,EAAE,OAAO,6BAA6B,CAAC;IACrE;GACF;GACA,MAAM,OAAO,OAAO,IAAI,KAAK,IAAI;GACjC,IAAI,CAAC,MAAM;IACT,eAAe,UAAU,KAAK,EAAE,OAAO,uBAAuB,KAAK,KAAK,GAAG,CAAC;IAC5E;GACF;GACA,SAAS;GACT,MAAM,SAAS,MAAM,KAAK,QAAQ,KAAK,MAAM,EAAE,OAAO,CAAC;GACvD,MAAM,UAAU,KAAK,UAAU,EAAE,OAAO,CAAC;GACzC,IAAI,OAAO,WAAW,OAAO,IAAI,OAAO,kBAAkB;IACxD,eAAe,UAAU,KAAK,EAAE,OAAO,gCAAgC,CAAC;IACxE;GACF;GACA,SAAS,UAAU,KAAK;IACtB,gBAAgB;IAChB,kBAAkB,OAAO,OAAO,WAAW,OAAO,CAAC;GACrD,CAAC;GACD,SAAS,IAAI,OAAO;EACtB,SAAS,OAAO;GACd,eAAe,UAAU,OAAO,UAAU,MAAM,KAAK,EACnD,OAAO,iBAAiB,QAAQ,MAAM,UAAU,OAAO,KAAK,EAC9D,CAAC;EACH;CACF;CAEA,eAAe,gBAA+B;EAC5C,KAAK,QAAQ,oBAAoB,SAAS,OAAO;EACjD,YAAY;EACZ,MAAM,gBAAgB,YAAY,MAAM;EACxC,OAAO,uBAAuB;EAC9B,KAAK,MAAM,cAAc,mBAAmB,WAAW,MAAM;EAC7D,MAAM,CAAC,gBAAgB,MAAM,QAAQ,WAAW,CAC9C,eACA,sBAAsB,cAAc,CACtC,CAAC;EACD,IAAI,kBAAkB,SAAS,KAAK,eAAe,SAAS,GAC1D,MAAM,IAAI,MAAM,iDAAiD;EAEnE,IAAI,cAAc,WAAW,YAAY,MAAM,aAAa;CAC9D;AACF;AAEA,SAAS,SAAS,SAA0B,iBAA2C;CACrF,OAAO,IAAI,SAAS,SAAS,WAAW;EACtC,IAAI,OAAO;EACX,MAAM,SAAmB,CAAC;EAC1B,QAAQ,GAAG,SAAS,UAAkB;GACpC,QAAQ,MAAM;GACd,IAAI,OAAO,iBAAiB;IAC1B,uBAAO,IAAI,MAAM,8BAA8B,CAAC;IAChD,QAAQ,QAAQ;IAChB;GACF;GACA,OAAO,KAAK,KAAK;EACnB,CAAC;EACD,QAAQ,GAAG,SAAS,MAAM;EAC1B,QAAQ,GAAG,aAAa;GACtB,IAAI;IACF,QAAQ,KAAK,MAAM,OAAO,OAAO,MAAM,CAAC,CAAC,SAAS,MAAM,CAAC,CAAC;GAC5D,SAAS,OAAO;IACd,OAAO,KAAK;GACd;EACF,CAAC;CACH,CAAC;AACH;AAEA,SAASA,WAAS,OAAkD;CAClE,OAAO,OAAO,UAAU,YAAY,UAAU,QAAQ,CAAC,MAAM,QAAQ,KAAK;AAC5E;;;AC1JA,MAAM,qBAAqB,KAAK;AAChC,MAAM,uBAAuB;AAC7B,MAAM,kCAAkC,KAAK,OAAO;AACpD,MAAM,mCAAmC,IAAI,OAAO;AACpD,MAAM,gCAAgC;AACtC,MAAM,qBAAqB;AAC3B,MAAM,gBAAgB;;AAEtB,MAAM,0BAA0B;;AAmDhC,SAAgB,yBAAyB,SAAyD;CAChG,cAAc,OAAO;CACrB,MAAM,kBAAkB,QAAQ,mBAAA;CAChC,IAAI,CAAC,OAAO,cAAc,eAAe,KAAK,mBAAmB,GAC/D,MAAM,IAAI,UAAU,0DAA0D;CAEhF,MAAM,iBAAiB,QAAQ,kBAAkB;CACjD,MAAM,qBAAqB,QAAQ,sBAAsB,kBAAkB;CAC3E,IAAI,CAAC,OAAO,cAAc,kBAAkB,KAAK,qBAAqB,GACpE,MAAM,IAAI,UAAU,iEAAiE;CAEvF,MAAM,YAAY,QAAQ,aAAa;CACvC,iBAAiB,WAAW,WAAW;CACvC,MAAM,aAAa,QAAQ,cAAc;CACzC,IAAI,CAAC,OAAO,SAAS,UAAU,KAAK,cAAc,GAChD,MAAM,IAAI,UAAU,iDAAiD;CAEvE,MAAM,UAAU,QAAQ,WAAW,gBAAgB,QAAQ,KAAK;CAChE,MAAM,uBAAuB,QAAQ,wBAAwB;CAC7D,MAAM,wBAAwB,QAAQ,yBAAyB;CAC/D,MAAM,wBAAwB,QAAQ,yBAAyB;CAC/D,MAAM,qBAAqB,QAAQ,sBAAsB;CACzD,mCACE;EACE;EACA,aAAa,QAAQ,oBAAoB;EACzC,iBAAiB;EACjB,kBAAkB;EAClB,2BAA2B;EAC3B,8BAA8B;EAC9B;EACA,kBAAkB;CACpB,GACA,uBACF;CACA,uCAAuC,QAAQ,iBAAiB,4BAA4B;CAC5F,iBAAiB,oBAAoB,oBAAoB;CAEzD,MAAM,SAAS,gBAAgB,QAAQ,MAAM;CAC7C,MAAM,gBAAgB,sCAAsC,QAAQ,MAAM;CAC1E,OAAO;EACL,IAAI;EACJ,aAAa;EACb,OAAO,QAAQ;EACf,SAAS;EACT,iBAAiB;GACf,eAAe;GACf,UAAU,QAAQ;GAClB,OAAO,QAAQ;GACf,SAAS,EAAE,GAAG,QAAQ;GACtB,cAAc;GACd,mBAAmB;GACnB,sBAAsB;GACtB,iBAAiB;GACjB,YAAY;GACZ,oBAAoB,QAAQ,oBAAoB;GAChD,mBAAmB;GACnB,oBAAoB;GACpB,0BAA0B;GAC1B,mBAAmB,QAAQ,mBAAmB;GAC9C,uBAAuB;GACvB,gBAAgB;GAChB,QAAQ,SAAS,oBAAoB;GACrC,gBAAgB,QAAQ,WAAW;EACrC;EACA,MAAM,QAAQ,SAAS;GACrB,MAAM,WAAW,MAAM,uBAAuB;IAC5C,OAAO,QAAQ;IACf,UAAU,QAAQ,OAAO;IACzB,GAAI,QAAQ,kBAAkB,EAAE,QAAQ,QAAQ,gBAAgB,IAAI,CAAC;IACrE,GAAI,QAAQ,SAAS,EAAE,QAAQ,QAAQ,OAAO,IAAI,CAAC;GACrD,CAAC;GACD,IAAI;GACJ,MAAM,kBAAgE,CAAC;GACvE,MAAM,SAAS,MAAM,eAAe;IAClC,OAAO;IACP,KAAK,YAAY;KACf,aAAa,MAAM,iCAAiC;MAClD,MAAM,QAAQ;MACd,SAAS,QAAQ;MACjB,kBAAkB,gBAAgB;OAChC,gBAAgB,KAAK,gBAAgB,WAAW,CAAC;OACjD,QAAQ,gBAAgB,WAAW;MACrC;MACA,OAAO,QAAQ;MACf,QAAQ;OACN;OACA,aAAa,wBAAwB,QAAQ,kBAAkB,QAAQ,MAAM;OAC7E,iBAAiB;OACjB,kBAAkB;OAClB,2BAA2B;OAC3B,8BAA8B;OAC9B,kBAAkB;OAClB;MACF;MACA,YAAY,QAAQ;MACpB,SAAS;MACT,OAAO,QAAQ;MACf,OAAO,QAAQ;MACf,GAAI,QAAQ,WAAW,EAAE,MAAM,QAAQ,SAAS,IAAI,CAAC;MACrD,GAAI,QAAQ,SAAS,EAAE,QAAQ,QAAQ,OAAO,IAAI,CAAC;KACrD,CAAC;KACD,QAAQ,MAAM,gCAAgC;MAC5C,QAAQ;MACR,OAAO,QAAQ;MACf,OAAO,QAAQ,MAAM,KAAK,SAAS,KAAK,IAAI;MAC5C,QAAQ,QAAQ;KAClB,CAAC;KAwCD,MAAM,SAAS,kBAAkB,MAvCf,4BAAqC;MACrD,OAAO;MACP,YAAY;MACZ,QAAQ;MACR,OAAO;OACL,WAAW;OACX,UAAU,QAAQ;OAClB,cAAc,QAAQ;OACtB,YAAY;QACV,SAAS,WAAW;QACpB,QAAQ,WAAW;QACnB,OAAO,QAAQ;QACf;OACF;OACA,cAAc;QACZ,KAAK,SAAS;QACd,OAAO,SAAS;QAChB,WAAW;OACb;OACA,WAAW,QAAQ,MAAM,KAAK,EAAE,MAAM,aAAa,kBAAkB;QACnE;QACA;QACA;OACF,EAAE;OACF;OACA,QAAQ;QACN,eAAe,QAAQ,OAAO;QAC9B,aAAa,QAAQ,OAAO;QAC5B,gBAAgB,QAAQ,OAAO;OACjC;OAGA,GAAI,QAAQ,aAAa,EAAE,YAAY,QAAQ,WAAW,IAAI,CAAC;MACjE;MACA,GAAI,SAAS,EAAE,OAAO,IAAI,CAAC;MAC3B,gBAAgB,CAAC,qBAAqB,OAAO,cAAc,aAAa,CAAC;MACzE;MACA,GAAI,QAAQ,SAAS,EAAE,QAAQ,QAAQ,OAAO,IAAI,CAAC;KACrD,CAAC,IACsC,OAAO,WAAW;MACvD,QAAQ,MAAM,yDAAyD;OACrE,QAAQ;OACR;OACA;MACF,CAAC;KACH,CAAC;KACD,MAAM,wBAAwB,WAAW,sBAAsB;KAC/D,MAAM,kBAAkB,WAAW,gBAAgB;KACnD,IAAI,OAAO,eAAe,uBACxB,MAAM,IAAI,MACR,qBAAqB,OAAO,WAAW,gDAAgD,uBACzF;KAEF,WAAW,wBAAwB;KACnC,OAAO;MACL,GAAG;MACH,WAAW,SAAS,MAAM;MAC1B,SAAS;OACP,GAAG,OAAO;OACV,sBAAsB;OACtB,4BAA4B;OAC5B;MACF;KACF;IACF;IACA,SAAS,YAAY;KAKnB,MAAM,UAAS,MAJO,QAAQ,WAAW,CACvC,GAAI,aAAa,CAAC,WAAW,MAAM,CAAC,IAAI,CAAC,GACzC,SAAS,MAAM,CACjB,CAAC,EAAA,CACsB,SAAS,UAC9B,MAAM,WAAW,aAAa,CAAC,MAAM,MAAM,IAAI,CAAC,CAClD;KACA,IAAI,OAAO,SAAS,GAClB,MAAM,IAAI,eAAe,QAAQ,kCAAkC;IAEvE;GACF,CAAC;GACD,QAAQ,MAAM,kCAAkC;IAC9C,QAAQ;IACR,aAAa,OAAO;IACpB,wBAAwB,OAAO,QAAQ;IACvC,YAAY,OAAO;IACnB,UAAU,OAAO,SAAS;GAC5B,CAAC;GACD,OAAO;EACT;CACF;AACF;AAEA,SAAS,kBACP,OACA,mBAC8C;CAC9C,IAAI,CAAC,SAAS,KAAK,GAAG,MAAM,IAAI,MAAM,0CAA0C;CAChF,IAAI,OAAO,MAAM,WAAW,YAAY,CAAC,MAAM,OAAO,KAAK,GACzD,MAAM,IAAI,MAAM,oCAAoC;CAMtD,MAAM,UAAU,sBAAsB,MAAM,QAAQ;CACpD,IAAI,QAAQ,kBAAkB,KAAA,GAC5B,MAAM,IAAI,MAAM,8CAA8C,QAAQ,eAAe;CAEvF,MAAM,WAAW,QAAQ;CACzB,MAAM,mBAAmB,QAAQ,SAAS;CAC1C,KAAK,MAAM,aAAa,QAAQ,UAC9B,kBACE,UAAU,OACV,GAAG,UAAU,OAAO,UAAU,OAAO,OAAO,UAAU,SAAS,GAAG,IAAI,UAAU,SAClF;CAEF,IAAI,CAAC,MAAM,QAAQ,MAAM,UAAU,GACjC,MAAM,IAAI,MAAM,6CAA6C;CAE/D,IAAI,CAAC,OAAO,cAAc,MAAM,UAAU,KAAM,MAAM,cAAyB,GAC7E,MAAM,IAAI,MAAM,4DAA4D;CAE9E,IAAI,CAAC,SAAS,MAAM,OAAO,GACzB,MAAM,IAAI,MAAM,2CAA2C;CAE7D,OAAO;EACL,QAAQ,MAAM;EACd;EACA,YAAY,MAAM;EAClB,YAAY,MAAM;EAClB,SAAS;GAAE,GAAG,MAAM;GAAS;EAAiB;CAChD;AACF;AAEA,SAAS,cAAc,SAA0C;CAC/D,KAAK,MAAM,CAAC,MAAM,UAAU,CAC1B,CAAC,WAAW,QAAQ,OAAO,GAC3B,CAAC,SAAS,QAAQ,KAAK,CACzB,GACE,IAAI,OAAO,UAAU,YAAY,CAAC,MAAM,KAAK,KAAK,UAAU,MAAM,KAAK,GACrE,MAAM,IAAI,UAAU,YAAY,KAAK,oCAAoC;CAG7E,IAAI,OAAO,QAAQ,SAAS,YAC1B,MAAM,IAAI,UAAU,kCAAkC;CAExD,IAAI,OAAO,QAAQ,oBAAoB,YACrC,MAAM,IAAI,UAAU,6CAA6C;CAEnE,IACE,QAAQ,qBAAqB,KAAA,MAC5B,CAAC,OAAO,cAAc,QAAQ,gBAAgB,KAAK,QAAQ,oBAAoB,IAEhF,MAAM,IAAI,UAAU,2DAA2D;AAEnF;AAEA,SAAS,wBACP,YACA,QACQ;CACR,IAAI,eAAe,KAAA,GAAW,OAAO;CACrC,MAAM,UAAU,OAAO,gBAAgB,OAAO,cAAc;CAC5D,IAAI,CAAC,OAAO,cAAc,OAAO,KAAK,WAAW,GAC/C,MAAM,IAAI,MAAM,iEAAiE;CAEnF,OAAO;AACT;AAEA,SAAS,iBAAiB,OAAe,OAAqB;CAC5D,IAAI,CAAC,OAAO,cAAc,KAAK,KAAK,SAAS,KAAK,QAAQ,oBACxD,MAAM,IAAI,UAAU,YAAY,MAAM,yBAAyB,oBAAoB;AAEvF;AAEA,SAAS,gBAAgB,OAAmC;CAC1D,MAAM,UAAU,oBAAoB,KAAK;CACzC,IAAI,CAAC,SACH,MAAM,IAAI,MACR,iCAAiC,MAAM,6CACzC;CAEF,OAAO;EACL,oBAAoB,QAAQ,QAAQ;EACpC,qBAAqB,QAAQ,SAAS;CACxC;AACF;AAEA,SAAS,gBACP,QAC4C;CAC5C,IAAI,CAAC,QAAQ,OAAO,KAAA;CACpB,OAAO;EACL,GAAI,OAAO,UAAU,EAAE,SAAS,OAAO,QAAQ,IAAI,CAAC;EACpD,GAAI,OAAO,OAAO,EAAE,MAAM,OAAO,KAAK,IAAI,CAAC;EAC3C,GAAI,OAAO,MAAM,EAAE,KAAK,4BAA4B,OAAO,GAAG,EAAE,IAAI,CAAC;EACrE,GAAI,OAAO,SAAS,EAAE,QAAQ,EAAE,GAAG,OAAO,OAAO,EAAE,IAAI,CAAC;CAC1D;AACF;AAEA,SAAS,SAAS,OAAkD;CAClE,OAAO,OAAO,UAAU,YAAY,UAAU,QAAQ,CAAC,MAAM,QAAQ,KAAK;AAC5E"}
|
|
1
|
+
{"version":3,"file":"dspy-rlm-engine-DqjER2sV.js","names":["isRecord"],"sources":["../src/analyst/finding-codec.ts","../src/analyst/trace-tool-callback.ts","../src/analyst/dspy-rlm-engine.ts"],"sourcesContent":["/**\n * The raw-finding wire contract — one decoder for both sides of the process\n * boundary.\n *\n * A finding array crosses a process boundary twice: an optimizer bridge writes\n * it in Python, TypeScript reads it back. Two decoders written independently\n * accept different shapes, so Python could report a completed investigation\n * whose rows TypeScript then dropped — the paid work disappearing between two\n * \"successes\". This module is the TypeScript half; `finding_contract.json`\n * carries the same rules to the Python half, generated from the schema here.\n *\n * Decoding never invents an empty result. A value that is not an array of rows\n * fails with the exact type it was, and a malformed row is reported with its\n * index and path while its valid siblings survive.\n */\n\nimport { type RawAnalystFinding, RawAnalystFindingSchema } from './finding-signature'\nimport { parseFindingSubject } from './finding-subject'\nimport { coerceJson } from './parse-tolerant'\n\n/**\n * Version of the wire contract both languages implement. Bump when the\n * accepted row shape changes; the Python package pins the same value, and the\n * emitted contract carries it so a mismatched pair is visible.\n */\nexport const FINDING_WIRE_CONTRACT_VERSION = 1\n\n/** Why one row was refused. Stable across languages: Python reports the same codes. */\nexport type FindingRejectionCode = 'not-an-object' | 'schema' | 'invalid-subject' | 'row-limit'\n\n/** A refused row, named precisely enough to repair or report. */\nexport interface RejectedFindingRow {\n /** Position in the submitted array. */\n readonly index: number\n /** Dotted path to the offending field, `''` for the row itself. */\n readonly path: string\n readonly code: FindingRejectionCode\n readonly message: string\n}\n\nexport interface DecodedFindingArray {\n readonly accepted: RawAnalystFinding[]\n readonly rejected: RejectedFindingRow[]\n /**\n * Set when the value was not a finding array at all — the array itself is\n * the failure, so `accepted` and `rejected` are both empty and a caller must\n * not read the result as \"no findings\".\n */\n readonly topLevelError?: string\n}\n\n/**\n * Rows past this count are refused rather than validated. A model that emits\n * thousands of rows has lost the plot, and the diagnostics for them would\n * dwarf the answer they came with.\n */\nexport const MAX_FINDING_ROWS = 500\n\n/**\n * Decode a submitted findings value into accepted rows plus per-row\n * diagnostics.\n *\n * Accepts an array of rows, or a string carrying one (a JSON array, possibly\n * fenced). Anything else — an object, a number, null, undefined — sets\n * `topLevelError` naming the type received: a caller that asked for findings\n * and got a number has a defect to report, not an empty result to record.\n */\nexport function decodeRawFindingArray(value: unknown): DecodedFindingArray {\n const rows = findingRows(value)\n if (typeof rows === 'string') return { accepted: [], rejected: [], topLevelError: rows }\n\n const accepted: RawAnalystFinding[] = []\n const rejected: RejectedFindingRow[] = []\n for (const [index, row] of rows.entries()) {\n if (index >= MAX_FINDING_ROWS) {\n rejected.push({\n index,\n path: '',\n code: 'row-limit',\n message: `findings array exceeds ${MAX_FINDING_ROWS} rows`,\n })\n continue\n }\n const decoded = decodeRow(row, index)\n if ('finding' in decoded) accepted.push(decoded.finding)\n else rejected.push(decoded.rejection)\n }\n return { accepted, rejected }\n}\n\nfunction decodeRow(\n row: unknown,\n index: number,\n): { finding: RawAnalystFinding } | { rejection: RejectedFindingRow } {\n if (row === null || typeof row !== 'object' || Array.isArray(row)) {\n return {\n rejection: {\n index,\n path: '',\n code: 'not-an-object',\n message: `finding row must be an object, received ${describe(row)}`,\n },\n }\n }\n const parsed = RawAnalystFindingSchema.safeParse(row)\n if (!parsed.success) {\n const issue = parsed.error.issues[0]\n const path = (issue?.path ?? []).join('.')\n // The schema refines `subject` against the grammar; name that refusal\n // separately so a repair turn can be told which rule the row broke.\n const code: FindingRejectionCode =\n path === 'subject' && parseFindingSubject((row as { subject?: string }).subject) === null\n ? 'invalid-subject'\n : 'schema'\n return {\n rejection: { index, path, code, message: issue?.message ?? 'row does not match the schema' },\n }\n }\n return { finding: parsed.data }\n}\n\n/**\n * Narrow a submitted value to candidate rows, or return the message naming why\n * it is not a findings array at all.\n */\nfunction findingRows(value: unknown): unknown[] | string {\n if (Array.isArray(value)) return value\n if (typeof value === 'string') {\n const parsed = coerceJson(value)\n if (parsed === undefined) {\n return 'findings must be a JSON array; the string received is not JSON'\n }\n if (Array.isArray(parsed)) return parsed\n const unwrapped = findingsProperty(parsed)\n if (unwrapped !== undefined) return unwrapped\n return `findings must be a JSON array, received ${describe(parsed)}`\n }\n const unwrapped = findingsProperty(value)\n if (unwrapped !== undefined) return unwrapped\n return `findings must be an array, received ${describe(value)}`\n}\n\n/** `{ findings: [...] }` is the one wrapper models emit often enough to unwrap. */\nfunction findingsProperty(value: unknown): unknown[] | undefined {\n if (value === null || typeof value !== 'object' || Array.isArray(value)) return undefined\n const inner = (value as Record<string, unknown>).findings\n return Array.isArray(inner) ? inner : undefined\n}\n\nfunction describe(value: unknown): string {\n if (value === null) return 'null'\n if (Array.isArray(value)) return 'an array'\n return `a ${typeof value}`\n}\n\n/** One line per refused row, for a repair prompt or an error message. */\nexport function describeRejectedRows(rejected: readonly RejectedFindingRow[]): string {\n return rejected\n .map((row) => `row ${row.index}${row.path ? ` field '${row.path}'` : ''}: ${row.message}`)\n .join('\\n')\n}\n","import { randomBytes } from 'node:crypto'\nimport { createServer, type IncomingMessage, type ServerResponse } from 'node:http'\nimport {\n type ExternalOptimizerCallbackLimits,\n resolveExternalOptimizerCallbackLimits,\n} from '../campaign/external-optimizer-contracts'\nimport {\n closeServer,\n listenLocal,\n sendJsonIfOpen,\n waitForActiveHandlers,\n} from '../campaign/external-optimizer-http'\nimport type { TraceAnalysisToolDescriptor } from '../trace-analyst/tools'\n\nexport interface TraceToolCallback {\n url: string\n token: string\n calls: () => number\n close: () => Promise<void>\n}\n\nexport type TraceToolCallbackLimits = ExternalOptimizerCallbackLimits\n\n/** Expose one bounded trace-tool set only on an authenticated loopback socket. */\nexport async function startTraceToolCallback(args: {\n tools: readonly TraceAnalysisToolDescriptor[]\n maxCalls: number\n /** Trace-tool request/response byte limits. Omitted fields use finite defaults. */\n limits?: Partial<TraceToolCallbackLimits>\n signal?: AbortSignal\n}): Promise<TraceToolCallback> {\n if (!Number.isSafeInteger(args.maxCalls) || args.maxCalls <= 0) {\n throw new TypeError('trace tool callback maxCalls must be a positive safe integer')\n }\n const limits = resolveExternalOptimizerCallbackLimits(args.limits, 'trace tool callback limits')\n args.signal?.throwIfAborted()\n const byName = new Map(args.tools.map((tool) => [tool.name, tool]))\n if (byName.size !== args.tools.length) {\n throw new Error('trace tool callback received duplicate tool names')\n }\n\n const token = randomBytes(32).toString('hex')\n let calls = 0\n let accepting = true\n let closePromise: Promise<void> | undefined\n const activeControllers = new Set<AbortController>()\n const activeHandlers = new Set<Promise<void>>()\n const server = createServer((request, response) => {\n if (!accepting) {\n sendJsonIfOpen(response, 503, { error: 'trace tool callback is closing' })\n return\n }\n const controller = new AbortController()\n const abortRequest = (): void => {\n request.destroy()\n response.destroy()\n }\n activeControllers.add(controller)\n controller.signal.addEventListener('abort', abortRequest, { once: true })\n\n let handler!: Promise<void>\n handler = handleRequest(request, response, controller.signal).finally(() => {\n controller.signal.removeEventListener('abort', abortRequest)\n activeControllers.delete(controller)\n activeHandlers.delete(handler)\n })\n activeHandlers.add(handler)\n void handler.catch(() => undefined)\n })\n const port = await listenLocal(server)\n const close = (): Promise<void> => {\n closePromise ??= closeCallback()\n return closePromise\n }\n const onAbort = (): void => {\n void close().catch(() => undefined)\n }\n args.signal?.addEventListener('abort', onAbort, { once: true })\n if (args.signal?.aborted) onAbort()\n\n return {\n url: `http://127.0.0.1:${port}/call`,\n token,\n calls: () => calls,\n close,\n }\n\n async function handleRequest(\n request: IncomingMessage,\n response: ServerResponse,\n signal: AbortSignal,\n ): Promise<void> {\n try {\n if (request.method !== 'POST' || request.url !== '/call') {\n sendJsonIfOpen(response, 404, { error: 'not found' })\n return\n }\n if (request.headers.authorization !== `Bearer ${token}`) {\n sendJsonIfOpen(response, 401, { error: 'unauthorized' })\n return\n }\n if (calls >= args.maxCalls) {\n sendJsonIfOpen(response, 429, { error: 'trace tool call limit reached' })\n return\n }\n const body = await readJson(request, limits.maxRequestBytes)\n if (!isRecord(body) || typeof body.name !== 'string' || !('args' in body)) {\n sendJsonIfOpen(response, 400, { error: 'name and args are required' })\n return\n }\n const tool = byName.get(body.name)\n if (!tool) {\n sendJsonIfOpen(response, 404, { error: `unknown trace tool '${body.name}'` })\n return\n }\n calls += 1\n const result = await tool.handler(body.args, { signal })\n const encoded = JSON.stringify({ result })\n if (Buffer.byteLength(encoded) > limits.maxResponseBytes) {\n sendJsonIfOpen(response, 413, { error: 'trace tool response too large' })\n return\n }\n response.writeHead(200, {\n 'content-type': 'application/json; charset=utf-8',\n 'content-length': String(Buffer.byteLength(encoded)),\n })\n response.end(encoded)\n } catch (error) {\n sendJsonIfOpen(response, signal.aborted ? 499 : 400, {\n error: error instanceof Error ? error.message : String(error),\n })\n }\n }\n\n async function closeCallback(): Promise<void> {\n args.signal?.removeEventListener('abort', onAbort)\n accepting = false\n const closingServer = closeServer(server)\n server.closeIdleConnections?.()\n for (const controller of activeControllers) controller.abort()\n const [serverResult] = await Promise.allSettled([\n closingServer,\n waitForActiveHandlers(activeHandlers),\n ])\n if (activeControllers.size !== 0 || activeHandlers.size !== 0) {\n throw new Error('trace tool callback closed with active requests')\n }\n if (serverResult?.status === 'rejected') throw serverResult.reason\n }\n}\n\nfunction readJson(request: IncomingMessage, maxRequestBytes: number): Promise<unknown> {\n return new Promise((resolve, reject) => {\n let size = 0\n const chunks: Buffer[] = []\n request.on('data', (chunk: Buffer) => {\n size += chunk.length\n if (size > maxRequestBytes) {\n reject(new Error('trace tool request too large'))\n request.destroy()\n return\n }\n chunks.push(chunk)\n })\n request.on('error', reject)\n request.on('end', () => {\n try {\n resolve(JSON.parse(Buffer.concat(chunks).toString('utf8')))\n } catch (error) {\n reject(error)\n }\n })\n })\n}\n\nfunction isRecord(value: unknown): value is Record<string, unknown> {\n return typeof value === 'object' && value !== null && !Array.isArray(value)\n}\n","import {\n assertExternalOptimizerModelBudget,\n type ExternalOptimizerModelCall,\n type ExternalOptimizerModelExecutionObservation,\n type ExternalOptimizerModelProxy,\n type ExternalOptimizerRunnerCommand,\n removeCredentialEnvironment,\n resolveExternalOptimizerCallbackLimits,\n resolveExternalOptimizerProcessLimits,\n} from '../campaign/external-optimizer-contracts'\nimport { startExternalOptimizerModelProxy } from '../campaign/external-optimizer-model-proxy'\nimport { runWithCleanup } from '../campaign/external-optimizer-resources'\nimport { runExternalOptimizerProcess } from '../campaign/external-optimizer-subprocess'\nimport type { CustomTokenPricing } from '../cost-ledger'\nimport { resolveModelPricing } from '../metrics'\nimport {\n DEFAULT_TRACE_ANALYST_OUTPUT_TOKENS,\n type TraceAnalysisEngine,\n type TraceAnalysisEngineResult,\n} from './engine'\nimport { decodeRawFindingArray } from './finding-codec'\nimport { startTraceToolCallback, type TraceToolCallbackLimits } from './trace-tool-callback'\n\nconst DEFAULT_TIMEOUT_MS = 10 * 60_000\nconst DEFAULT_MAX_COST_USD = 1\nconst DEFAULT_MAX_MODEL_REQUEST_BYTES = 16 * 1024 * 1024\nconst DEFAULT_MAX_MODEL_RESPONSE_BYTES = 4 * 1024 * 1024\nconst DEFAULT_TRACE_TOOL_TIMEOUT_MS = 60_000\nconst MAX_TIMER_DELAY_MS = 2_147_483_647\nconst BRIDGE_MODULE = 'agent_eval_rpc.dspy_rlm_bridge'\n/** Bumped whenever this engine's execution behavior changes. */\nconst DSPY_RLM_ENGINE_VERSION = '1.0.0'\n\nexport interface DspyRlmTraceEngineOptions {\n /** Caller-owned execution path. Agent Eval never receives provider credentials. */\n call: ExternalOptimizerModelCall\n /** Stable public identity for the caller-owned path, such as an AgentProfile digest. */\n callRef: string\n /** Persist the finite execution record returned for every admitted call. */\n recordExecution: (observation: ExternalOptimizerModelExecutionObservation) => void\n model: string\n /** Exact provider rates. Required when the model is absent from the pricing table. */\n pricing?: CustomTokenPricing\n /** Maximum provider spend for one investigation. Default: 1 USD. */\n maxCostUsd?: number\n /** Controller response cap. Default: {@link DEFAULT_TRACE_ANALYST_OUTPUT_TOKENS}. */\n maxOutputTokens?: number\n /**\n * Thinking tokens one controller turn may bill on top of its completion.\n * A reasoning model bills these beyond `maxOutputTokens`, so the cost\n * reservation must cover them. Default: four times the completion cap.\n */\n maxReasoningTokens?: number\n /** Maximum caller-owned model invocations. Default derives from the analysis limits. */\n maxModelRequests?: number\n /** Maximum model request bytes. Default: 16 MiB. */\n maxModelRequestBytes?: number\n /** Maximum model response bytes. Default: 4 MiB. */\n maxModelResponseBytes?: number\n /** Deadline for one caller-owned model invocation. Default: the whole analysis deadline. */\n modelRequestTimeoutMs?: number\n /** Trace-tool loopback request/response byte limits. */\n traceToolLimits?: Partial<TraceToolCallbackLimits>\n /** Deadline for one Python-to-Node trace-tool call. Default: 60 seconds. */\n traceToolTimeoutMs?: number\n /**\n * How the controller's reasoning and code fields are obtained.\n *\n * `tolerant` parses marker output strictly first, then recovers the fields\n * deterministically from prose plus a fenced code block — the shape coding\n * models naturally emit — at no extra model cost. `two-step` extracts with a\n * second call per turn. `chat` accepts marker output only. Default:\n * `tolerant`.\n */\n controlAdapter?: 'chat' | 'two-step' | 'tolerant'\n /** Python command used to load agent-eval-rpc[dspy]. Default: python. */\n runner?: ExternalOptimizerRunnerCommand\n /** Whole investigation deadline. Default: 10 minutes. */\n timeoutMs?: number\n}\n\n/** Use the official DSPy RLM as a bounded recursive trace-analysis engine. */\nexport function createDspyRlmTraceEngine(options: DspyRlmTraceEngineOptions): TraceAnalysisEngine {\n assertOptions(options)\n const maxOutputTokens = options.maxOutputTokens ?? DEFAULT_TRACE_ANALYST_OUTPUT_TOKENS\n if (!Number.isSafeInteger(maxOutputTokens) || maxOutputTokens <= 0) {\n throw new TypeError('DSPy RLM maxOutputTokens must be a positive safe integer')\n }\n const controlAdapter = options.controlAdapter ?? 'tolerant'\n const maxReasoningTokens = options.maxReasoningTokens ?? maxOutputTokens * 4\n if (!Number.isSafeInteger(maxReasoningTokens) || maxReasoningTokens < 0) {\n throw new TypeError('DSPy RLM maxReasoningTokens must be a non-negative safe integer')\n }\n const timeoutMs = options.timeoutMs ?? DEFAULT_TIMEOUT_MS\n assertTimerDelay(timeoutMs, 'timeoutMs')\n const maxCostUsd = options.maxCostUsd ?? DEFAULT_MAX_COST_USD\n if (!Number.isFinite(maxCostUsd) || maxCostUsd <= 0) {\n throw new TypeError('DSPy RLM maxCostUsd must be positive and finite')\n }\n const pricing = options.pricing ?? pricingForModel(options.model)\n const maxModelRequestBytes = options.maxModelRequestBytes ?? DEFAULT_MAX_MODEL_REQUEST_BYTES\n const maxModelResponseBytes = options.maxModelResponseBytes ?? DEFAULT_MAX_MODEL_RESPONSE_BYTES\n const modelRequestTimeoutMs = options.modelRequestTimeoutMs ?? timeoutMs\n const traceToolTimeoutMs = options.traceToolTimeoutMs ?? DEFAULT_TRACE_TOOL_TIMEOUT_MS\n assertExternalOptimizerModelBudget(\n {\n maxCostUsd,\n maxRequests: options.maxModelRequests ?? 1,\n maxRequestBytes: maxModelRequestBytes,\n maxResponseBytes: maxModelResponseBytes,\n maxOutputTokensPerRequest: maxOutputTokens,\n maxReasoningTokensPerRequest: maxReasoningTokens,\n pricing,\n requestTimeoutMs: modelRequestTimeoutMs,\n },\n 'DSPy RLM model limits',\n )\n resolveExternalOptimizerCallbackLimits(options.traceToolLimits, 'DSPy RLM trace tool limits')\n assertTimerDelay(traceToolTimeoutMs, 'traceToolTimeoutMs')\n\n const runner = sanitizedRunner(options.runner)\n const processLimits = resolveExternalOptimizerProcessLimits(runner?.limits)\n return {\n id: 'dspy-rlm',\n description: 'Official DSPy RLM with bounded trace tools and metered model calls.',\n model: options.model,\n version: DSPY_RLM_ENGINE_VERSION,\n executionConfig: {\n bridge_module: BRIDGE_MODULE,\n call_ref: options.callRef,\n model: options.model,\n pricing: { ...pricing },\n max_cost_usd: maxCostUsd,\n max_output_tokens: maxOutputTokens,\n max_reasoning_tokens: maxReasoningTokens,\n control_adapter: controlAdapter,\n timeout_ms: timeoutMs,\n max_model_requests: options.maxModelRequests ?? null,\n max_request_bytes: maxModelRequestBytes,\n max_response_bytes: maxModelResponseBytes,\n model_request_timeout_ms: modelRequestTimeoutMs,\n trace_tool_limits: options.traceToolLimits ?? null,\n trace_tool_timeout_ms: traceToolTimeoutMs,\n process_limits: processLimits,\n runner: runner ? 'caller-supplied' : 'default',\n runner_command: runner?.command ?? null,\n },\n async analyze(request) {\n const callback = await startTraceToolCallback({\n tools: request.tools,\n maxCalls: request.limits.maxToolCalls,\n ...(options.traceToolLimits ? { limits: options.traceToolLimits } : {}),\n ...(request.signal ? { signal: request.signal } : {}),\n })\n let modelProxy: ExternalOptimizerModelProxy | undefined\n const modelExecutions: ExternalOptimizerModelExecutionObservation[] = []\n const result = await runWithCleanup({\n label: 'DSPy RLM trace-analysis resources',\n run: async () => {\n modelProxy = await startExternalOptimizerModelProxy({\n call: options.call,\n callRef: options.callRef,\n recordExecution: (observation) => {\n modelExecutions.push(structuredClone(observation))\n options.recordExecution(observation)\n },\n model: options.model,\n budget: {\n maxCostUsd,\n maxRequests: resolveMaxModelRequests(options.maxModelRequests, request.limits),\n maxRequestBytes: maxModelRequestBytes,\n maxResponseBytes: maxModelResponseBytes,\n maxOutputTokensPerRequest: maxOutputTokens,\n maxReasoningTokensPerRequest: maxReasoningTokens,\n requestTimeoutMs: modelRequestTimeoutMs,\n pricing,\n },\n costLedger: request.costLedger,\n channel: 'analyst',\n phase: request.costPhase,\n actor: request.analystId,\n ...(request.costTags ? { tags: request.costTags } : {}),\n ...(request.signal ? { signal: request.signal } : {}),\n })\n request.log?.('trace analyst engine started', {\n engine: 'dspy-rlm',\n model: options.model,\n tools: request.tools.map((tool) => tool.name),\n limits: request.limits,\n })\n const raw = await runExternalOptimizerProcess<unknown>({\n label: 'DSPy RLM trace analysis',\n tempPrefix: 'agent-eval-dspy-rlm-',\n module: BRIDGE_MODULE,\n input: {\n operation: 'analyze',\n question: request.question,\n instructions: request.instructions,\n modelProxy: {\n baseUrl: modelProxy.baseUrl,\n apiKey: modelProxy.apiKey,\n model: options.model,\n maxOutputTokens,\n },\n toolCallback: {\n url: callback.url,\n token: callback.token,\n timeoutMs: traceToolTimeoutMs,\n },\n toolSpecs: request.tools.map(({ name, description, parameters }) => ({\n name,\n description,\n parameters,\n })),\n controlAdapter,\n limits: {\n maxIterations: request.limits.maxIterations,\n maxLlmCalls: request.limits.maxLlmCalls,\n maxOutputChars: request.limits.maxOutputChars,\n },\n // Omitted entirely when the caller supplied none, so a request\n // without structured inputs sends the payload it always sent.\n ...(request.taskInputs ? { taskInputs: request.taskInputs } : {}),\n },\n ...(runner ? { runner } : {}),\n additionalArgs: ['--max-input-bytes', String(processLimits.maxInputBytes)],\n timeoutMs,\n ...(request.signal ? { signal: request.signal } : {}),\n })\n const parsed = parseBridgeOutput(raw, (index, reason) => {\n request.log?.('finding rejected: bridge row failed schema validation', {\n engine: 'dspy-rlm',\n index,\n reason,\n })\n })\n const successfulCompletions = modelProxy.successfulCompletions()\n const requestAttempts = modelProxy.requestAttempts()\n if (parsed.modelCalls !== successfulCompletions) {\n throw new Error(\n `DSPy RLM reported ${parsed.modelCalls} model calls, but the provider proxy recorded ${successfulCompletions}`,\n )\n }\n modelProxy.assertExecutionComplete()\n return {\n ...parsed,\n toolCalls: callback.calls(),\n runtime: {\n ...parsed.runtime,\n modelRequestAttempts: requestAttempts,\n modelSuccessfulCompletions: successfulCompletions,\n modelExecutions,\n },\n } satisfies TraceAnalysisEngineResult\n },\n cleanup: async () => {\n const results = await Promise.allSettled([\n ...(modelProxy ? [modelProxy.close()] : []),\n callback.close(),\n ])\n const errors = results.flatMap((entry) =>\n entry.status === 'rejected' ? [entry.reason] : [],\n )\n if (errors.length > 0) {\n throw new AggregateError(errors, 'DSPy RLM resource cleanup failed')\n }\n },\n })\n request.log?.('trace analyst engine completed', {\n engine: 'dspy-rlm',\n model_calls: result.modelCalls,\n model_request_attempts: result.runtime.modelRequestAttempts,\n tool_calls: result.toolCalls,\n findings: result.findings.length,\n })\n return result\n },\n }\n}\n\nfunction parseBridgeOutput(\n value: unknown,\n onRejectedFinding: (index: number, reason: string) => void,\n): Omit<TraceAnalysisEngineResult, 'toolCalls'> {\n if (!isRecord(value)) throw new Error('DSPy RLM bridge output must be an object')\n if (typeof value.answer !== 'string' || !value.answer.trim()) {\n throw new Error('DSPy RLM bridge returned no answer')\n }\n // Findings are model output: one malformed row is model noise, not a bridge\n // fault, and the rest of the paid investigation must survive it. The codec\n // is the same decoder the Python side runs, so a row this bridge accepts is\n // a row the optimizer could report, and vice versa.\n const decoded = decodeRawFindingArray(value.findings)\n if (decoded.topLevelError !== undefined) {\n throw new Error(`DSPy RLM bridge findings must be an array: ${decoded.topLevelError}`)\n }\n const findings = decoded.accepted\n const rejectedFindings = decoded.rejected.length\n for (const rejection of decoded.rejected) {\n onRejectedFinding(\n rejection.index,\n `${rejection.code}${rejection.path ? ` at ${rejection.path}` : ''}: ${rejection.message}`,\n )\n }\n if (!Array.isArray(value.trajectory)) {\n throw new Error('DSPy RLM bridge trajectory must be an array')\n }\n if (!Number.isSafeInteger(value.modelCalls) || (value.modelCalls as number) <= 0) {\n throw new Error('DSPy RLM bridge modelCalls must be a positive safe integer')\n }\n if (!isRecord(value.runtime)) {\n throw new Error('DSPy RLM bridge runtime must be an object')\n }\n return {\n answer: value.answer,\n findings,\n trajectory: value.trajectory,\n modelCalls: value.modelCalls as number,\n runtime: { ...value.runtime, rejectedFindings },\n }\n}\n\nfunction assertOptions(options: DspyRlmTraceEngineOptions): void {\n for (const [name, value] of [\n ['callRef', options.callRef],\n ['model', options.model],\n ] as const) {\n if (typeof value !== 'string' || !value.trim() || value !== value.trim()) {\n throw new TypeError(`DSPy RLM ${name} must be a trimmed non-empty string`)\n }\n }\n if (typeof options.call !== 'function') {\n throw new TypeError('DSPy RLM call must be a function')\n }\n if (typeof options.recordExecution !== 'function') {\n throw new TypeError('DSPy RLM recordExecution must be a function')\n }\n if (\n options.maxModelRequests !== undefined &&\n (!Number.isSafeInteger(options.maxModelRequests) || options.maxModelRequests <= 0)\n ) {\n throw new TypeError('DSPy RLM maxModelRequests must be a positive safe integer')\n }\n}\n\nfunction resolveMaxModelRequests(\n configured: number | undefined,\n limits: { maxIterations: number; maxLlmCalls: number },\n): number {\n if (configured !== undefined) return configured\n const derived = limits.maxIterations + limits.maxLlmCalls + 1\n if (!Number.isSafeInteger(derived) || derived <= 0) {\n throw new Error('DSPy RLM analysis limits produce an invalid model request limit')\n }\n return derived\n}\n\nfunction assertTimerDelay(value: number, field: string): void {\n if (!Number.isSafeInteger(value) || value <= 0 || value > MAX_TIMER_DELAY_MS) {\n throw new TypeError(`DSPy RLM ${field} must be between 1 and ${MAX_TIMER_DELAY_MS}`)\n }\n}\n\nfunction pricingForModel(model: string): CustomTokenPricing {\n const pricing = resolveModelPricing(model)\n if (!pricing) {\n throw new Error(\n `no pricing is configured for '${model}'; provide DspyRlmTraceEngineOptions.pricing`,\n )\n }\n return {\n inputUsdPerMillion: pricing.input * 1_000,\n outputUsdPerMillion: pricing.output * 1_000,\n }\n}\n\nfunction sanitizedRunner(\n runner: ExternalOptimizerRunnerCommand | undefined,\n): ExternalOptimizerRunnerCommand | undefined {\n if (!runner) return undefined\n return {\n ...(runner.command ? { command: runner.command } : {}),\n ...(runner.args ? { args: runner.args } : {}),\n ...(runner.env ? { env: removeCredentialEnvironment(runner.env) } : {}),\n ...(runner.limits ? { limits: { ...runner.limits } } : {}),\n }\n}\n\nfunction isRecord(value: unknown): value is Record<string, unknown> {\n return typeof value === 'object' && value !== null && !Array.isArray(value)\n}\n"],"mappings":";;;;;;;;;;;;;;AAmEA,SAAgB,sBAAsB,OAAqC;CACzE,MAAM,OAAO,YAAY,KAAK;CAC9B,IAAI,OAAO,SAAS,UAAU,OAAO;EAAE,UAAU,CAAC;EAAG,UAAU,CAAC;EAAG,eAAe;CAAK;CAEvF,MAAM,WAAgC,CAAC;CACvC,MAAM,WAAiC,CAAC;CACxC,KAAK,MAAM,CAAC,OAAO,QAAQ,KAAK,QAAQ,GAAG;EACzC,IAAI,SAAA,KAA2B;GAC7B,SAAS,KAAK;IACZ;IACA,MAAM;IACN,MAAM;IACN,SAAS;GACX,CAAC;GACD;EACF;EACA,MAAM,UAAU,UAAU,KAAK,KAAK;EACpC,IAAI,aAAa,SAAS,SAAS,KAAK,QAAQ,OAAO;OAClD,SAAS,KAAK,QAAQ,SAAS;CACtC;CACA,OAAO;EAAE;EAAU;CAAS;AAC9B;AAEA,SAAS,UACP,KACA,OACoE;CACpE,IAAI,QAAQ,QAAQ,OAAO,QAAQ,YAAY,MAAM,QAAQ,GAAG,GAC9D,OAAO,EACL,WAAW;EACT;EACA,MAAM;EACN,MAAM;EACN,SAAS,2CAA2C,SAAS,GAAG;CAClE,EACF;CAEF,MAAM,SAAS,wBAAwB,UAAU,GAAG;CACpD,IAAI,CAAC,OAAO,SAAS;EACnB,MAAM,QAAQ,OAAO,MAAM,OAAO;EAClC,MAAM,QAAQ,OAAO,QAAQ,CAAC,EAAA,CAAG,KAAK,GAAG;EAOzC,OAAO,EACL,WAAW;GAAE;GAAO;GAAM,MAJ1B,SAAS,aAAa,oBAAqB,IAA6B,OAAO,MAAM,OACjF,oBACA;GAE4B,SAAS,OAAO,WAAW;EAAgC,EAC7F;CACF;CACA,OAAO,EAAE,SAAS,OAAO,KAAK;AAChC;;;;;AAMA,SAAS,YAAY,OAAoC;CACvD,IAAI,MAAM,QAAQ,KAAK,GAAG,OAAO;CACjC,IAAI,OAAO,UAAU,UAAU;EAC7B,MAAM,SAAS,WAAW,KAAK;EAC/B,IAAI,WAAW,KAAA,GACb,OAAO;EAET,IAAI,MAAM,QAAQ,MAAM,GAAG,OAAO;EAClC,MAAM,YAAY,iBAAiB,MAAM;EACzC,IAAI,cAAc,KAAA,GAAW,OAAO;EACpC,OAAO,2CAA2C,SAAS,MAAM;CACnE;CACA,MAAM,YAAY,iBAAiB,KAAK;CACxC,IAAI,cAAc,KAAA,GAAW,OAAO;CACpC,OAAO,uCAAuC,SAAS,KAAK;AAC9D;;AAGA,SAAS,iBAAiB,OAAuC;CAC/D,IAAI,UAAU,QAAQ,OAAO,UAAU,YAAY,MAAM,QAAQ,KAAK,GAAG,OAAO,KAAA;CAChF,MAAM,QAAS,MAAkC;CACjD,OAAO,MAAM,QAAQ,KAAK,IAAI,QAAQ,KAAA;AACxC;AAEA,SAAS,SAAS,OAAwB;CACxC,IAAI,UAAU,MAAM,OAAO;CAC3B,IAAI,MAAM,QAAQ,KAAK,GAAG,OAAO;CACjC,OAAO,KAAK,OAAO;AACrB;;;;ACjIA,eAAsB,uBAAuB,MAMd;CAC7B,IAAI,CAAC,OAAO,cAAc,KAAK,QAAQ,KAAK,KAAK,YAAY,GAC3D,MAAM,IAAI,UAAU,8DAA8D;CAEpF,MAAM,SAAS,uCAAuC,KAAK,QAAQ,4BAA4B;CAC/F,KAAK,QAAQ,eAAe;CAC5B,MAAM,SAAS,IAAI,IAAI,KAAK,MAAM,KAAK,SAAS,CAAC,KAAK,MAAM,IAAI,CAAC,CAAC;CAClE,IAAI,OAAO,SAAS,KAAK,MAAM,QAC7B,MAAM,IAAI,MAAM,mDAAmD;CAGrE,MAAM,QAAQ,YAAY,EAAE,CAAC,CAAC,SAAS,KAAK;CAC5C,IAAI,QAAQ;CACZ,IAAI,YAAY;CAChB,IAAI;CACJ,MAAM,oCAAoB,IAAI,IAAqB;CACnD,MAAM,iCAAiB,IAAI,IAAmB;CAC9C,MAAM,SAAS,cAAc,SAAS,aAAa;EACjD,IAAI,CAAC,WAAW;GACd,eAAe,UAAU,KAAK,EAAE,OAAO,iCAAiC,CAAC;GACzE;EACF;EACA,MAAM,aAAa,IAAI,gBAAgB;EACvC,MAAM,qBAA2B;GAC/B,QAAQ,QAAQ;GAChB,SAAS,QAAQ;EACnB;EACA,kBAAkB,IAAI,UAAU;EAChC,WAAW,OAAO,iBAAiB,SAAS,cAAc,EAAE,MAAM,KAAK,CAAC;EAExE,IAAI;EACJ,UAAU,cAAc,SAAS,UAAU,WAAW,MAAM,CAAC,CAAC,cAAc;GAC1E,WAAW,OAAO,oBAAoB,SAAS,YAAY;GAC3D,kBAAkB,OAAO,UAAU;GACnC,eAAe,OAAO,OAAO;EAC/B,CAAC;EACD,eAAe,IAAI,OAAO;EAC1B,QAAa,YAAY,KAAA,CAAS;CACpC,CAAC;CACD,MAAM,OAAO,MAAM,YAAY,MAAM;CACrC,MAAM,cAA6B;EACjC,iBAAiB,cAAc;EAC/B,OAAO;CACT;CACA,MAAM,gBAAsB;EAC1B,MAAW,CAAC,CAAC,YAAY,KAAA,CAAS;CACpC;CACA,KAAK,QAAQ,iBAAiB,SAAS,SAAS,EAAE,MAAM,KAAK,CAAC;CAC9D,IAAI,KAAK,QAAQ,SAAS,QAAQ;CAElC,OAAO;EACL,KAAK,oBAAoB,KAAK;EAC9B;EACA,aAAa;EACb;CACF;CAEA,eAAe,cACb,SACA,UACA,QACe;EACf,IAAI;GACF,IAAI,QAAQ,WAAW,UAAU,QAAQ,QAAQ,SAAS;IACxD,eAAe,UAAU,KAAK,EAAE,OAAO,YAAY,CAAC;IACpD;GACF;GACA,IAAI,QAAQ,QAAQ,kBAAkB,UAAU,SAAS;IACvD,eAAe,UAAU,KAAK,EAAE,OAAO,eAAe,CAAC;IACvD;GACF;GACA,IAAI,SAAS,KAAK,UAAU;IAC1B,eAAe,UAAU,KAAK,EAAE,OAAO,gCAAgC,CAAC;IACxE;GACF;GACA,MAAM,OAAO,MAAM,SAAS,SAAS,OAAO,eAAe;GAC3D,IAAI,CAACA,WAAS,IAAI,KAAK,OAAO,KAAK,SAAS,YAAY,EAAE,UAAU,OAAO;IACzE,eAAe,UAAU,KAAK,EAAE,OAAO,6BAA6B,CAAC;IACrE;GACF;GACA,MAAM,OAAO,OAAO,IAAI,KAAK,IAAI;GACjC,IAAI,CAAC,MAAM;IACT,eAAe,UAAU,KAAK,EAAE,OAAO,uBAAuB,KAAK,KAAK,GAAG,CAAC;IAC5E;GACF;GACA,SAAS;GACT,MAAM,SAAS,MAAM,KAAK,QAAQ,KAAK,MAAM,EAAE,OAAO,CAAC;GACvD,MAAM,UAAU,KAAK,UAAU,EAAE,OAAO,CAAC;GACzC,IAAI,OAAO,WAAW,OAAO,IAAI,OAAO,kBAAkB;IACxD,eAAe,UAAU,KAAK,EAAE,OAAO,gCAAgC,CAAC;IACxE;GACF;GACA,SAAS,UAAU,KAAK;IACtB,gBAAgB;IAChB,kBAAkB,OAAO,OAAO,WAAW,OAAO,CAAC;GACrD,CAAC;GACD,SAAS,IAAI,OAAO;EACtB,SAAS,OAAO;GACd,eAAe,UAAU,OAAO,UAAU,MAAM,KAAK,EACnD,OAAO,iBAAiB,QAAQ,MAAM,UAAU,OAAO,KAAK,EAC9D,CAAC;EACH;CACF;CAEA,eAAe,gBAA+B;EAC5C,KAAK,QAAQ,oBAAoB,SAAS,OAAO;EACjD,YAAY;EACZ,MAAM,gBAAgB,YAAY,MAAM;EACxC,OAAO,uBAAuB;EAC9B,KAAK,MAAM,cAAc,mBAAmB,WAAW,MAAM;EAC7D,MAAM,CAAC,gBAAgB,MAAM,QAAQ,WAAW,CAC9C,eACA,sBAAsB,cAAc,CACtC,CAAC;EACD,IAAI,kBAAkB,SAAS,KAAK,eAAe,SAAS,GAC1D,MAAM,IAAI,MAAM,iDAAiD;EAEnE,IAAI,cAAc,WAAW,YAAY,MAAM,aAAa;CAC9D;AACF;AAEA,SAAS,SAAS,SAA0B,iBAA2C;CACrF,OAAO,IAAI,SAAS,SAAS,WAAW;EACtC,IAAI,OAAO;EACX,MAAM,SAAmB,CAAC;EAC1B,QAAQ,GAAG,SAAS,UAAkB;GACpC,QAAQ,MAAM;GACd,IAAI,OAAO,iBAAiB;IAC1B,uBAAO,IAAI,MAAM,8BAA8B,CAAC;IAChD,QAAQ,QAAQ;IAChB;GACF;GACA,OAAO,KAAK,KAAK;EACnB,CAAC;EACD,QAAQ,GAAG,SAAS,MAAM;EAC1B,QAAQ,GAAG,aAAa;GACtB,IAAI;IACF,QAAQ,KAAK,MAAM,OAAO,OAAO,MAAM,CAAC,CAAC,SAAS,MAAM,CAAC,CAAC;GAC5D,SAAS,OAAO;IACd,OAAO,KAAK;GACd;EACF,CAAC;CACH,CAAC;AACH;AAEA,SAASA,WAAS,OAAkD;CAClE,OAAO,OAAO,UAAU,YAAY,UAAU,QAAQ,CAAC,MAAM,QAAQ,KAAK;AAC5E;;;AC1JA,MAAM,qBAAqB,KAAK;AAChC,MAAM,uBAAuB;AAC7B,MAAM,kCAAkC,KAAK,OAAO;AACpD,MAAM,mCAAmC,IAAI,OAAO;AACpD,MAAM,gCAAgC;AACtC,MAAM,qBAAqB;AAC3B,MAAM,gBAAgB;;AAEtB,MAAM,0BAA0B;;AAmDhC,SAAgB,yBAAyB,SAAyD;CAChG,cAAc,OAAO;CACrB,MAAM,kBAAkB,QAAQ,mBAAA;CAChC,IAAI,CAAC,OAAO,cAAc,eAAe,KAAK,mBAAmB,GAC/D,MAAM,IAAI,UAAU,0DAA0D;CAEhF,MAAM,iBAAiB,QAAQ,kBAAkB;CACjD,MAAM,qBAAqB,QAAQ,sBAAsB,kBAAkB;CAC3E,IAAI,CAAC,OAAO,cAAc,kBAAkB,KAAK,qBAAqB,GACpE,MAAM,IAAI,UAAU,iEAAiE;CAEvF,MAAM,YAAY,QAAQ,aAAa;CACvC,iBAAiB,WAAW,WAAW;CACvC,MAAM,aAAa,QAAQ,cAAc;CACzC,IAAI,CAAC,OAAO,SAAS,UAAU,KAAK,cAAc,GAChD,MAAM,IAAI,UAAU,iDAAiD;CAEvE,MAAM,UAAU,QAAQ,WAAW,gBAAgB,QAAQ,KAAK;CAChE,MAAM,uBAAuB,QAAQ,wBAAwB;CAC7D,MAAM,wBAAwB,QAAQ,yBAAyB;CAC/D,MAAM,wBAAwB,QAAQ,yBAAyB;CAC/D,MAAM,qBAAqB,QAAQ,sBAAsB;CACzD,mCACE;EACE;EACA,aAAa,QAAQ,oBAAoB;EACzC,iBAAiB;EACjB,kBAAkB;EAClB,2BAA2B;EAC3B,8BAA8B;EAC9B;EACA,kBAAkB;CACpB,GACA,uBACF;CACA,uCAAuC,QAAQ,iBAAiB,4BAA4B;CAC5F,iBAAiB,oBAAoB,oBAAoB;CAEzD,MAAM,SAAS,gBAAgB,QAAQ,MAAM;CAC7C,MAAM,gBAAgB,sCAAsC,QAAQ,MAAM;CAC1E,OAAO;EACL,IAAI;EACJ,aAAa;EACb,OAAO,QAAQ;EACf,SAAS;EACT,iBAAiB;GACf,eAAe;GACf,UAAU,QAAQ;GAClB,OAAO,QAAQ;GACf,SAAS,EAAE,GAAG,QAAQ;GACtB,cAAc;GACd,mBAAmB;GACnB,sBAAsB;GACtB,iBAAiB;GACjB,YAAY;GACZ,oBAAoB,QAAQ,oBAAoB;GAChD,mBAAmB;GACnB,oBAAoB;GACpB,0BAA0B;GAC1B,mBAAmB,QAAQ,mBAAmB;GAC9C,uBAAuB;GACvB,gBAAgB;GAChB,QAAQ,SAAS,oBAAoB;GACrC,gBAAgB,QAAQ,WAAW;EACrC;EACA,MAAM,QAAQ,SAAS;GACrB,MAAM,WAAW,MAAM,uBAAuB;IAC5C,OAAO,QAAQ;IACf,UAAU,QAAQ,OAAO;IACzB,GAAI,QAAQ,kBAAkB,EAAE,QAAQ,QAAQ,gBAAgB,IAAI,CAAC;IACrE,GAAI,QAAQ,SAAS,EAAE,QAAQ,QAAQ,OAAO,IAAI,CAAC;GACrD,CAAC;GACD,IAAI;GACJ,MAAM,kBAAgE,CAAC;GACvE,MAAM,SAAS,MAAM,eAAe;IAClC,OAAO;IACP,KAAK,YAAY;KACf,aAAa,MAAM,iCAAiC;MAClD,MAAM,QAAQ;MACd,SAAS,QAAQ;MACjB,kBAAkB,gBAAgB;OAChC,gBAAgB,KAAK,gBAAgB,WAAW,CAAC;OACjD,QAAQ,gBAAgB,WAAW;MACrC;MACA,OAAO,QAAQ;MACf,QAAQ;OACN;OACA,aAAa,wBAAwB,QAAQ,kBAAkB,QAAQ,MAAM;OAC7E,iBAAiB;OACjB,kBAAkB;OAClB,2BAA2B;OAC3B,8BAA8B;OAC9B,kBAAkB;OAClB;MACF;MACA,YAAY,QAAQ;MACpB,SAAS;MACT,OAAO,QAAQ;MACf,OAAO,QAAQ;MACf,GAAI,QAAQ,WAAW,EAAE,MAAM,QAAQ,SAAS,IAAI,CAAC;MACrD,GAAI,QAAQ,SAAS,EAAE,QAAQ,QAAQ,OAAO,IAAI,CAAC;KACrD,CAAC;KACD,QAAQ,MAAM,gCAAgC;MAC5C,QAAQ;MACR,OAAO,QAAQ;MACf,OAAO,QAAQ,MAAM,KAAK,SAAS,KAAK,IAAI;MAC5C,QAAQ,QAAQ;KAClB,CAAC;KAwCD,MAAM,SAAS,kBAAkB,MAvCf,4BAAqC;MACrD,OAAO;MACP,YAAY;MACZ,QAAQ;MACR,OAAO;OACL,WAAW;OACX,UAAU,QAAQ;OAClB,cAAc,QAAQ;OACtB,YAAY;QACV,SAAS,WAAW;QACpB,QAAQ,WAAW;QACnB,OAAO,QAAQ;QACf;OACF;OACA,cAAc;QACZ,KAAK,SAAS;QACd,OAAO,SAAS;QAChB,WAAW;OACb;OACA,WAAW,QAAQ,MAAM,KAAK,EAAE,MAAM,aAAa,kBAAkB;QACnE;QACA;QACA;OACF,EAAE;OACF;OACA,QAAQ;QACN,eAAe,QAAQ,OAAO;QAC9B,aAAa,QAAQ,OAAO;QAC5B,gBAAgB,QAAQ,OAAO;OACjC;OAGA,GAAI,QAAQ,aAAa,EAAE,YAAY,QAAQ,WAAW,IAAI,CAAC;MACjE;MACA,GAAI,SAAS,EAAE,OAAO,IAAI,CAAC;MAC3B,gBAAgB,CAAC,qBAAqB,OAAO,cAAc,aAAa,CAAC;MACzE;MACA,GAAI,QAAQ,SAAS,EAAE,QAAQ,QAAQ,OAAO,IAAI,CAAC;KACrD,CAAC,IACsC,OAAO,WAAW;MACvD,QAAQ,MAAM,yDAAyD;OACrE,QAAQ;OACR;OACA;MACF,CAAC;KACH,CAAC;KACD,MAAM,wBAAwB,WAAW,sBAAsB;KAC/D,MAAM,kBAAkB,WAAW,gBAAgB;KACnD,IAAI,OAAO,eAAe,uBACxB,MAAM,IAAI,MACR,qBAAqB,OAAO,WAAW,gDAAgD,uBACzF;KAEF,WAAW,wBAAwB;KACnC,OAAO;MACL,GAAG;MACH,WAAW,SAAS,MAAM;MAC1B,SAAS;OACP,GAAG,OAAO;OACV,sBAAsB;OACtB,4BAA4B;OAC5B;MACF;KACF;IACF;IACA,SAAS,YAAY;KAKnB,MAAM,UAAS,MAJO,QAAQ,WAAW,CACvC,GAAI,aAAa,CAAC,WAAW,MAAM,CAAC,IAAI,CAAC,GACzC,SAAS,MAAM,CACjB,CAAC,EAAA,CACsB,SAAS,UAC9B,MAAM,WAAW,aAAa,CAAC,MAAM,MAAM,IAAI,CAAC,CAClD;KACA,IAAI,OAAO,SAAS,GAClB,MAAM,IAAI,eAAe,QAAQ,kCAAkC;IAEvE;GACF,CAAC;GACD,QAAQ,MAAM,kCAAkC;IAC9C,QAAQ;IACR,aAAa,OAAO;IACpB,wBAAwB,OAAO,QAAQ;IACvC,YAAY,OAAO;IACnB,UAAU,OAAO,SAAS;GAC5B,CAAC;GACD,OAAO;EACT;CACF;AACF;AAEA,SAAS,kBACP,OACA,mBAC8C;CAC9C,IAAI,CAAC,SAAS,KAAK,GAAG,MAAM,IAAI,MAAM,0CAA0C;CAChF,IAAI,OAAO,MAAM,WAAW,YAAY,CAAC,MAAM,OAAO,KAAK,GACzD,MAAM,IAAI,MAAM,oCAAoC;CAMtD,MAAM,UAAU,sBAAsB,MAAM,QAAQ;CACpD,IAAI,QAAQ,kBAAkB,KAAA,GAC5B,MAAM,IAAI,MAAM,8CAA8C,QAAQ,eAAe;CAEvF,MAAM,WAAW,QAAQ;CACzB,MAAM,mBAAmB,QAAQ,SAAS;CAC1C,KAAK,MAAM,aAAa,QAAQ,UAC9B,kBACE,UAAU,OACV,GAAG,UAAU,OAAO,UAAU,OAAO,OAAO,UAAU,SAAS,GAAG,IAAI,UAAU,SAClF;CAEF,IAAI,CAAC,MAAM,QAAQ,MAAM,UAAU,GACjC,MAAM,IAAI,MAAM,6CAA6C;CAE/D,IAAI,CAAC,OAAO,cAAc,MAAM,UAAU,KAAM,MAAM,cAAyB,GAC7E,MAAM,IAAI,MAAM,4DAA4D;CAE9E,IAAI,CAAC,SAAS,MAAM,OAAO,GACzB,MAAM,IAAI,MAAM,2CAA2C;CAE7D,OAAO;EACL,QAAQ,MAAM;EACd;EACA,YAAY,MAAM;EAClB,YAAY,MAAM;EAClB,SAAS;GAAE,GAAG,MAAM;GAAS;EAAiB;CAChD;AACF;AAEA,SAAS,cAAc,SAA0C;CAC/D,KAAK,MAAM,CAAC,MAAM,UAAU,CAC1B,CAAC,WAAW,QAAQ,OAAO,GAC3B,CAAC,SAAS,QAAQ,KAAK,CACzB,GACE,IAAI,OAAO,UAAU,YAAY,CAAC,MAAM,KAAK,KAAK,UAAU,MAAM,KAAK,GACrE,MAAM,IAAI,UAAU,YAAY,KAAK,oCAAoC;CAG7E,IAAI,OAAO,QAAQ,SAAS,YAC1B,MAAM,IAAI,UAAU,kCAAkC;CAExD,IAAI,OAAO,QAAQ,oBAAoB,YACrC,MAAM,IAAI,UAAU,6CAA6C;CAEnE,IACE,QAAQ,qBAAqB,KAAA,MAC5B,CAAC,OAAO,cAAc,QAAQ,gBAAgB,KAAK,QAAQ,oBAAoB,IAEhF,MAAM,IAAI,UAAU,2DAA2D;AAEnF;AAEA,SAAS,wBACP,YACA,QACQ;CACR,IAAI,eAAe,KAAA,GAAW,OAAO;CACrC,MAAM,UAAU,OAAO,gBAAgB,OAAO,cAAc;CAC5D,IAAI,CAAC,OAAO,cAAc,OAAO,KAAK,WAAW,GAC/C,MAAM,IAAI,MAAM,iEAAiE;CAEnF,OAAO;AACT;AAEA,SAAS,iBAAiB,OAAe,OAAqB;CAC5D,IAAI,CAAC,OAAO,cAAc,KAAK,KAAK,SAAS,KAAK,QAAQ,oBACxD,MAAM,IAAI,UAAU,YAAY,MAAM,yBAAyB,oBAAoB;AAEvF;AAEA,SAAS,gBAAgB,OAAmC;CAC1D,MAAM,UAAU,oBAAoB,KAAK;CACzC,IAAI,CAAC,SACH,MAAM,IAAI,MACR,iCAAiC,MAAM,6CACzC;CAEF,OAAO;EACL,oBAAoB,QAAQ,QAAQ;EACpC,qBAAqB,QAAQ,SAAS;CACxC;AACF;AAEA,SAAS,gBACP,QAC4C;CAC5C,IAAI,CAAC,QAAQ,OAAO,KAAA;CACpB,OAAO;EACL,GAAI,OAAO,UAAU,EAAE,SAAS,OAAO,QAAQ,IAAI,CAAC;EACpD,GAAI,OAAO,OAAO,EAAE,MAAM,OAAO,KAAK,IAAI,CAAC;EAC3C,GAAI,OAAO,MAAM,EAAE,KAAK,4BAA4B,OAAO,GAAG,EAAE,IAAI,CAAC;EACrE,GAAI,OAAO,SAAS,EAAE,QAAQ,EAAE,GAAG,OAAO,OAAO,EAAE,IAAI,CAAC;CAC1D;AACF;AAEA,SAAS,SAAS,OAAkD;CAClE,OAAO,OAAO,UAAU,YAAY,UAAU,QAAQ,CAAC,MAAM,QAAQ,KAAK;AAC5E"}
|
|
@@ -1,7 +1,6 @@
|
|
|
1
|
-
import {
|
|
2
|
-
import {
|
|
3
|
-
import {
|
|
4
|
-
import { i as researchReport } from "./summary-report-Bgh8CpNK.js";
|
|
1
|
+
import { f as verifyAgentProfileCell, m as hashJson, s as buildAgentProfileCell } from "./agent-profile-cell-0gSi5ffD.js";
|
|
2
|
+
import { c as validateRunRecord } from "./run-record-DQpSf7t-.js";
|
|
3
|
+
import { i as researchReport } from "./summary-report-B16xy9Kd.js";
|
|
5
4
|
import { t as TraceEmitter } from "./emitter-DeQHiDMm.js";
|
|
6
5
|
import { t as FileSystemRawProviderSink } from "./raw-provider-sink-BQd7mzyT.js";
|
|
7
6
|
import { n as assertRunCaptured, t as RunIntegrityError } from "./integrity-Cy9WHAtb.js";
|
|
@@ -336,4 +335,4 @@ function defaultRunId(params) {
|
|
|
336
335
|
//#endregion
|
|
337
336
|
export { runEvalCampaign as t };
|
|
338
337
|
|
|
339
|
-
//# sourceMappingURL=eval-campaign-
|
|
338
|
+
//# sourceMappingURL=eval-campaign-Cs-7MiCs.js.map
|
|
@@ -1 +1 @@
|
|
|
1
|
-
{"version":3,"file":"eval-campaign-BeAjdhzC.js","names":[],"sources":["../src/eval-campaign.ts"],"sourcesContent":["/**\n * EvalCampaign — opinionated matrix runner that wires the four\n * capture-integrity directives by construction.\n *\n * The canonical benchmark shape — matrix runner → for each\n * (variant, scenario, seed) → start a TraceEmitter → call LLMs → end the\n * run → analyze — has a bug class at the integration boundary: raw\n * events not captured, route silently wrong, integrity not asserted,\n * analyst never run. The directives in `SKILL.md § Capture integrity`\n * are the mitigations.\n *\n * `EvalCampaign` is the structural fix — consumers don't wire the\n * integrity surface themselves; the campaign owns it. Specifically:\n *\n * - constructs a per-run `TraceStore` and `RawProviderSink` via factories\n * - builds the run's `ChatClient` through `chatFactory`, handing it the\n * run's raw sink and trace context — a transport built any other way has\n * no raw HTTP envelope, and `assertRunCaptured` says so\n * - calls `assertRunCaptured` after every `endRun` and routes failures\n * through a configurable policy (`throw` / `mark_failed` / `log`)\n * - assembles per-run `RunRecord`s and runs `researchReport` at the end\n * so the campaign artifact is launch-decision-grade by default\n * - embeds the campaign fingerprint (a SHA-256 over the canonicalised\n * run set) and optional `preregistrationHash` in the report\n *\n * The runner contract is intentionally narrow: produce a `CampaignRunOutcome`\n * given a fully-wired `CampaignRunContext`. Everything orchestration-shaped\n * lives in the campaign. This is the inversion-of-control point — consumers\n * stop writing matrix runners and start writing scenario-runners.\n *\n * Out of scope for v1 (tracked in `docs/research-report-methodology.md`):\n *\n * - Distributed/cluster execution (concurrency is local async)\n * - Adaptive sampling / sequential interim looks\n * - Resume from partial state across crashes\n * - LLM-call retry beyond what the caller's transport already does\n */\n\nimport {\n type AgentProfileCell,\n type AgentProfileCellInput,\n buildAgentProfileCell,\n verifyAgentProfileCell,\n} from './agent-profile-cell'\nimport type { ChatClient } from './analyst/chat-client'\nimport { hashJson } from './pre-registration'\nimport type {\n JudgeScoresRecord,\n RunCostProvenance,\n RunJudgeMetadata,\n RunOutcome,\n RunRecord,\n RunSplitTag,\n RunTaskFailure,\n RunTokenUsage,\n} from './run-record'\nimport { validateRunRecord } from './run-record'\nimport { type ResearchReport, type ResearchReportOptions, researchReport } from './summary-report'\nimport type { RunCompleteHook } from './trace/emitter'\nimport { TraceEmitter } from './trace/emitter'\nimport {\n assertRunCaptured,\n RunIntegrityError,\n type RunIntegrityExpectations,\n type RunIntegrityReport,\n} from './trace/integrity'\nimport { FileSystemRawProviderSink, type RawProviderSink } from './trace/raw-provider-sink'\nimport type { TraceStore } from './trace/store'\n\n// ── Public types ─────────────────────────────────────────────────────────\n\nexport interface CampaignVariant<V> {\n id: string\n payload: V\n}\n\nexport interface CampaignScenario {\n scenarioId: string\n /** Free-form metadata propagated to runs and reports. */\n tags?: Record<string, string>\n}\n\nexport interface CampaignRunContext<V> {\n /** Stable run id. The campaign generates this; the runner does not. */\n runId: string\n /** Logical experiment id (campaignId by default; overridable per-run via opts). */\n experimentId: string\n variant: V\n variantId: string\n scenarioId: string\n scenarioTags: Record<string, string>\n seed: number\n splitTag: RunSplitTag\n /**\n * The TraceEmitter for this run, with `onRunComplete` hooks pre-wired\n * (analyst auto-execution if configured, plus integrity check). The\n * runner MUST call `emitter.startRun` before doing any work and either\n * `emitter.endRun` or `emitter.abortRun` before returning.\n */\n emitter: TraceEmitter\n store: TraceStore\n rawSink: RawProviderSink\n /**\n * The run's model transport, built by `chatFactory` with this run's\n * `rawSink` and `runId` already bound.\n */\n chat: ChatClient\n}\n\n/** What the campaign binds into the run's transport. */\nexport interface CampaignChatWiring {\n /**\n * Raw provider sink for this run. Bind it into the transport: the campaign's\n * integrity check requires every LLM span to carry a matching raw request\n * event, so a transport built without it fails `assertRunCaptured`.\n */\n rawSink: RawProviderSink\n runId: string\n}\n\ninterface CampaignRunOutcomeFields {\n /** Did the run pass? Mirrors `RunOutcome.pass` semantics. */\n pass: boolean\n /** Score for the run on its split. Maps to `searchScore` or `holdoutScore`. */\n score: number\n /** Cost in USD, or null when the runner could not capture it. */\n costUsd: number | null\n /** Source of the cost amount. */\n costProvenance: RunCostProvenance\n tokenUsage: RunTokenUsage\n /** Snapshot model id (e.g. `claude-sonnet-4-6@2025-04-15`). */\n model: string\n /** sha256 of the effective prompt sent to the model. */\n promptHash: string\n /** sha256 of the effective config (model, temperature, tools, judges, splits). */\n configHash: string\n /** Optional extra numeric metrics to land in `outcome.raw`. */\n raw?: Record<string, number>\n /** Optional judge metadata when a judge was used. */\n judgeMetadata?: RunJudgeMetadata\n /**\n * Optional per-judge / per-dim breakdown for ensemble-judged runs.\n * Propagated to `outcome.judgeScores` on the resulting `RunRecord`.\n * Single-judge or scalar-only runs leave this unset.\n */\n judgeScores?: JudgeScoresRecord\n /**\n * Agent profile cell observed by the runner. When supplied, it overrides\n * `EvalCampaignOptions.agentProfile` for this run and must match the\n * outcome's `model` and `promptHash`.\n */\n agentProfile?: AgentProfileCell | AgentProfileCellInput\n}\n\n/** Campaign result with the same task-failure invariant as `RunRecord`. */\nexport type CampaignRunOutcome = CampaignRunOutcomeFields & RunTaskFailure\n\nexport type CampaignRunner<V> = (ctx: CampaignRunContext<V>) => Promise<CampaignRunOutcome>\n\nexport type CampaignIntegrityPolicy = 'throw' | 'mark_failed' | 'log'\n\nexport interface EvalCampaignOptions<V> {\n /**\n * Stable id for the campaign. Used as the default `experimentId` on\n * every run, and folded into the campaign fingerprint.\n */\n campaignId: string\n variants: CampaignVariant<V>[]\n scenarios: CampaignScenario[]\n /** Default `[0, 1, 2]`. */\n seeds?: number[]\n /** Default `'holdout'` — the split that anchors a launch decision. */\n splitTag?: RunSplitTag\n /** Git SHA the campaign is run against. Mandatory; `RunRecord` rejects unset. */\n commitSha: string\n /**\n * Build the model transport for one run. agent-eval executes no paid model:\n * the caller owns the transport and the credential never enters this\n * package. The campaign calls this once per run and passes the run's raw\n * provider sink and `runId`, so a transport that binds them captures the\n * raw HTTP envelope `assertRunCaptured` checks for.\n */\n chatFactory: (wiring: CampaignChatWiring) => ChatClient\n /**\n * Caller-declared identity of the execution route, folded into the campaign\n * fingerprint so two campaigns run against different endpoints do not share\n * one identity. agent-eval no longer knows the endpoint; the owner of\n * execution names it.\n */\n executionRef?: string\n /**\n * Per-run TraceStore factory. Common shape: a fresh store per run keyed\n * on `runId`. Implementations that share a store across the campaign\n * are valid — the campaign only writes through `emitter`.\n */\n storeFactory: (params: CampaignFactoryParams) => TraceStore\n /**\n * Per-run RawProviderSink factory. Defaults to `FileSystemRawProviderSink`\n * rooted at `${workDir}/raw-events/${runId}` if `workDir` is supplied;\n * otherwise required. Forensic capture is non-negotiable in a campaign\n * run — pass `NoopRawProviderSink` explicitly if you want to opt out.\n */\n rawSinkFactory?: (params: CampaignFactoryParams) => RawProviderSink\n /**\n * Filesystem root for default `rawSinkFactory`. Ignored if\n * `rawSinkFactory` is supplied.\n */\n workDir?: string\n /**\n * Extra `onRunComplete` hooks the campaign appends (after its own\n * integrity-check hook). Pass `traceAnalystOnRunComplete(...)` here.\n */\n onRunComplete?: RunCompleteHook[]\n /**\n * Per-run integrity expectations. Defaults to:\n * `{ llmSpansMin: 1, requireRawCoverageOfLlmSpans: true, requireOutcome: true }`.\n * Override (e.g. `{ llmSpansMin: 0 }`) for runs that don't call LLMs.\n */\n integrity?: RunIntegrityExpectations\n /** Behaviour when integrity fails. Default `'mark_failed'`. */\n onIntegrityFailure?: CampaignIntegrityPolicy\n /**\n * Per-run runner. Receives a fully-wired context; produces an outcome\n * the campaign converts into a `RunRecord`.\n */\n runner: CampaignRunner<V>\n /**\n * If set, the campaign computes `researchReport` at the end. `comparator`\n * is a `variantId`. Other fields are forwarded verbatim.\n */\n report?: { comparator?: string } & Omit<\n ResearchReportOptions,\n 'comparator' | 'preregistrationHash' | 'generatedAt'\n >\n /**\n * Hash of a signed `HypothesisManifest` (see `pre-registration.ts`).\n * Embedded in the campaign fingerprint and the research report.\n */\n preregistrationHash?: string\n /** Local concurrency. Default `1` (sequential). */\n concurrency?: number\n /**\n * Override the time source. Tests pass a mock to make wallMs deterministic.\n */\n now?: () => number\n /** Override the runId generator. Tests pin this. */\n runId?: (params: CampaignFactoryParams) => string\n /**\n * Agent profile cell for campaign runs. Static profiles can pass an object;\n * routers or variant-specific harnesses can pass a factory. The campaign\n * stamps the built cell onto every `RunRecord` and rejects profile/model or\n * profile/prompt contradictions.\n */\n agentProfile?:\n | AgentProfileCell\n | AgentProfileCellInput\n | ((\n params: CampaignFactoryParams & {\n variant: V\n scenarioTags: Record<string, string>\n },\n ) =>\n | AgentProfileCell\n | AgentProfileCellInput\n | Promise<AgentProfileCell | AgentProfileCellInput>)\n}\n\nexport interface CampaignFactoryParams {\n campaignId: string\n runId: string\n variantId: string\n scenarioId: string\n seed: number\n}\n\nexport interface FailedRun {\n runId: string\n variantId: string\n scenarioId: string\n seed: number\n reason: string\n error?: string\n}\n\nexport interface EvalCampaignResult {\n campaignId: string\n /** SHA-256 over canonicalised `(variantIds, scenarioIds, seeds, comparator, splitTag, executionRef, preregistrationHash)`. */\n campaignFingerprint: string\n preregistrationHash: string | null\n /** Successful runs only. Failed runs land in `failedRuns`. */\n runs: RunRecord[]\n /** Integrity reports for every successful run. */\n integrityReports: RunIntegrityReport[]\n failedRuns: FailedRun[]\n /** Computed when `report` is set on options. */\n report?: ResearchReport\n startedAt: string\n endedAt: string\n}\n\n// ── Implementation ───────────────────────────────────────────────────────\n\nconst DEFAULT_INTEGRITY: RunIntegrityExpectations = {\n llmSpansMin: 1,\n requireRawCoverageOfLlmSpans: true,\n requireOutcome: true,\n}\n\nexport async function runEvalCampaign<V>(\n opts: EvalCampaignOptions<V>,\n): Promise<EvalCampaignResult> {\n // ── Preflight ──────────────────────────────────────────────────────\n if (opts.variants.length === 0) {\n throw new Error('runEvalCampaign: variants must be non-empty.')\n }\n if (opts.scenarios.length === 0) {\n throw new Error('runEvalCampaign: scenarios must be non-empty.')\n }\n const variantIds = new Set<string>()\n for (const v of opts.variants) {\n if (variantIds.has(v.id)) {\n throw new Error(`runEvalCampaign: duplicate variant id \"${v.id}\".`)\n }\n variantIds.add(v.id)\n }\n const scenarioIds = new Set<string>()\n for (const s of opts.scenarios) {\n if (scenarioIds.has(s.scenarioId)) {\n throw new Error(`runEvalCampaign: duplicate scenarioId \"${s.scenarioId}\".`)\n }\n scenarioIds.add(s.scenarioId)\n }\n if (opts.report?.comparator && !variantIds.has(opts.report.comparator)) {\n throw new Error(\n `runEvalCampaign: report.comparator \"${opts.report.comparator}\" is not a configured variantId.`,\n )\n }\n if (!opts.commitSha) {\n throw new Error('runEvalCampaign: commitSha is required (every RunRecord needs it).')\n }\n\n const seeds = opts.seeds ?? [0, 1, 2]\n const splitTag: RunSplitTag = opts.splitTag ?? 'holdout'\n const concurrency = Math.max(1, opts.concurrency ?? 1)\n const integrity = { ...DEFAULT_INTEGRITY, ...(opts.integrity ?? {}) }\n const onIntegrityFailure: CampaignIntegrityPolicy = opts.onIntegrityFailure ?? 'mark_failed'\n const now = opts.now ?? (() => Date.now())\n const executionRef = opts.executionRef ?? null\n const preregistrationHash = opts.preregistrationHash ?? null\n\n const rawSinkFactory = opts.rawSinkFactory ?? defaultRawSinkFactory(opts.workDir)\n\n // ── Fingerprint ────────────────────────────────────────────────────\n const campaignFingerprint = await hashJson({\n campaignId: opts.campaignId,\n variants: opts.variants.map((v) => v.id).sort(),\n scenarios: opts.scenarios.map((s) => s.scenarioId).sort(),\n seeds: [...seeds].sort((a, b) => a - b),\n splitTag,\n comparator: opts.report?.comparator ?? null,\n executionRef,\n preregistrationHash,\n })\n\n // ── Plan the matrix ────────────────────────────────────────────────\n type Cell = { variant: CampaignVariant<V>; scenario: CampaignScenario; seed: number }\n const cells: Cell[] = []\n for (const variant of opts.variants) {\n for (const scenario of opts.scenarios) {\n for (const seed of seeds) {\n cells.push({ variant, scenario, seed })\n }\n }\n }\n\n const startedAt = new Date(now()).toISOString()\n const runs: RunRecord[] = []\n const integrityReports: RunIntegrityReport[] = []\n const failedRuns: FailedRun[] = []\n\n // ── Execute (bounded-concurrency worker pool) ──────────────────────\n // A genuine (non-CellExecutionError) error from any worker is a bug, not a\n // run-level failure. We must NOT let `Promise.all` reject mid-flight and\n // orphan the other workers' in-progress runs (emitters never finalized,\n // sinks/handles leak, partial work silently discarded). Instead: capture the\n // first genuine error, stop dispatching new cells so in-flight workers wind\n // down, finalize every still-open run, then re-throw the aggregated error.\n let cursor = 0\n let aborting = false\n const genuineErrors: unknown[] = []\n // Emitters for runs that are currently mid-flight, keyed by runId. A worker\n // registers its emitter before invoking the runner and removes it once the\n // run is finalized (via endRun/abortRun, success or failure). Anything left\n // here after the pool settles is an orphan we must abort.\n const openRuns = new Map<string, TraceEmitter>()\n\n async function worker(): Promise<void> {\n while (!aborting) {\n const i = cursor++\n if (i >= cells.length) return\n const cell = cells[i]!\n try {\n const result = await runOneCell(cell)\n runs.push(result.record)\n integrityReports.push(result.integrity)\n } catch (err) {\n if (err instanceof CellExecutionError) {\n failedRuns.push(err.failed)\n if (err.integrity) integrityReports.push(err.integrity)\n } else {\n // Genuine bug — not a runner failure, not an integrity failure.\n // Capture it and stop dispatching so peers wind down gracefully;\n // re-thrown after the pool settles. Do not surface here — that would\n // reject Promise.all and orphan the other workers' runs.\n genuineErrors.push(err)\n aborting = true\n return\n }\n }\n }\n }\n\n async function runOneCell(\n cell: Cell,\n ): Promise<{ record: RunRecord; integrity: RunIntegrityReport }> {\n const runId = (opts.runId ?? defaultRunId)({\n campaignId: opts.campaignId,\n runId: '', // unused by default generator\n variantId: cell.variant.id,\n scenarioId: cell.scenario.scenarioId,\n seed: cell.seed,\n })\n const factoryParams: CampaignFactoryParams = {\n campaignId: opts.campaignId,\n runId,\n variantId: cell.variant.id,\n scenarioId: cell.scenario.scenarioId,\n seed: cell.seed,\n }\n const store = opts.storeFactory(factoryParams)\n const rawSink = rawSinkFactory(factoryParams)\n\n const emitter = new TraceEmitter(store, {\n runId,\n now: opts.now,\n onRunComplete: opts.onRunComplete,\n })\n // Track this run as open so a genuine error elsewhere in the pool can\n // finalize it instead of orphaning it. Removed in the finally below.\n openRuns.set(runId, emitter)\n\n const chat = opts.chatFactory({ rawSink, runId })\n\n const ctx: CampaignRunContext<V> = {\n runId,\n experimentId: opts.campaignId,\n variant: cell.variant.payload,\n variantId: cell.variant.id,\n scenarioId: cell.scenario.scenarioId,\n scenarioTags: cell.scenario.tags ?? {},\n seed: cell.seed,\n splitTag,\n emitter,\n store,\n rawSink,\n chat,\n }\n\n try {\n const wallStart = now()\n let outcome: CampaignRunOutcome\n try {\n outcome = await opts.runner(ctx)\n } catch (err) {\n const message = err instanceof Error ? err.message : String(err)\n // The runner threw mid-execution. Abort the run so the emitter\n // finalizes. The only benign abortRun failure is \"the runner never\n // started the run\" (nothing to finalize). A store-write failure (disk\n // full, FS error) is a genuine diagnostic — surface it rather than\n // masking it as a plain runner failure.\n await finalizeAbort(emitter, runId, message)\n throw new CellExecutionError({\n runId,\n variantId: cell.variant.id,\n scenarioId: cell.scenario.scenarioId,\n seed: cell.seed,\n reason: 'runner_threw',\n error: message,\n })\n }\n const wallMs = now() - wallStart\n\n const integrityReport = await assertRunCaptured(store, runId, { ...integrity, rawSink })\n if (!integrityReport.ok) {\n switch (onIntegrityFailure) {\n case 'throw':\n throw new RunIntegrityError(integrityReport)\n case 'mark_failed':\n throw new CellExecutionError(\n {\n runId,\n variantId: cell.variant.id,\n scenarioId: cell.scenario.scenarioId,\n seed: cell.seed,\n reason: 'integrity_failed',\n error: integrityReport.issues.map((i) => i.code).join(', '),\n },\n integrityReport,\n )\n case 'log':\n // Caller wants the run admitted with a flagged report; fall through.\n break\n }\n }\n\n const recordOutcome: RunOutcome = {\n raw: outcome.raw ?? {},\n }\n if (splitTag === 'holdout') recordOutcome.holdoutScore = outcome.score\n else recordOutcome.searchScore = outcome.score\n if (outcome.judgeScores !== undefined) recordOutcome.judgeScores = outcome.judgeScores\n\n const record: RunRecord = {\n runId,\n experimentId: opts.campaignId,\n candidateId: cell.variant.id,\n seed: cell.seed,\n model: outcome.model,\n promptHash: outcome.promptHash,\n configHash: outcome.configHash,\n commitSha: opts.commitSha,\n wallMs,\n costUsd: outcome.costUsd,\n costProvenance: outcome.costProvenance,\n tokenUsage: outcome.tokenUsage,\n terminalOutcome: 'succeeded',\n judgeMetadata: outcome.judgeMetadata,\n outcome: recordOutcome,\n ...(outcome.failureClass ? { failureClass: outcome.failureClass } : {}),\n failureMode: outcome.failureMode,\n splitTag,\n scenarioId: cell.scenario.scenarioId,\n }\n const profileSource =\n outcome.agentProfile ??\n (typeof opts.agentProfile === 'function'\n ? await opts.agentProfile({\n campaignId: opts.campaignId,\n runId,\n variantId: cell.variant.id,\n scenarioId: cell.scenario.scenarioId,\n seed: cell.seed,\n variant: cell.variant.payload,\n scenarioTags: cell.scenario.tags ?? {},\n })\n : opts.agentProfile)\n if (profileSource !== undefined) {\n const agentProfile = await resolveAgentProfileCell(profileSource)\n assertAgentProfileMatchesRun(agentProfile, outcome.model, outcome.promptHash)\n record.agentProfile = agentProfile\n }\n return { record: validateRunRecord(record), integrity: integrityReport }\n } finally {\n // This run's worker has finished with it (success, run-level failure, or\n // genuine error). It is no longer the pool's job to finalize — drop it so\n // the post-settle sweep doesn't double-abort a finalized run.\n openRuns.delete(runId)\n }\n }\n\n const workers = Array.from({ length: Math.min(concurrency, cells.length) }, () => worker())\n // allSettled (not all): a genuine error in one worker must not reject the\n // pool mid-flight and orphan the others. Each worker captures its own\n // genuine error into `genuineErrors` and returns; we re-throw below.\n await Promise.allSettled(workers)\n\n // Finalize any run still open after the pool wound down. With the\n // stop-dispatch flag these are the runs that were mid-flight in peer workers\n // when the first genuine error fired — abort them so their emitters finalize\n // (hooks fire, store records a terminal status) instead of leaking.\n for (const [runId, emitter] of openRuns) {\n await finalizeAbort(emitter, runId, 'campaign aborted: genuine error in a sibling run')\n }\n openRuns.clear()\n\n if (genuineErrors.length > 0) {\n throw genuineErrors.length === 1\n ? genuineErrors[0]\n : new AggregateError(\n genuineErrors,\n `runEvalCampaign: ${genuineErrors.length} runs failed with genuine (non-run-level) errors`,\n )\n }\n\n // ── Optional research report ───────────────────────────────────────\n let report: ResearchReport | undefined\n if (opts.report) {\n const reportOpts: ResearchReportOptions = {\n ...opts.report,\n comparator: opts.report.comparator,\n split: splitTag === 'dev' ? 'search' : splitTag,\n generatedAt: new Date(now()).toISOString(),\n preregistrationHash: preregistrationHash ?? undefined,\n }\n report = await researchReport(runs, reportOpts)\n }\n\n const endedAt = new Date(now()).toISOString()\n\n return {\n campaignId: opts.campaignId,\n campaignFingerprint,\n preregistrationHash,\n runs,\n integrityReports,\n failedRuns,\n report,\n startedAt,\n endedAt,\n }\n}\n\n// ── Internal ─────────────────────────────────────────────────────────────\n\nclass CellExecutionError extends Error {\n readonly failed: FailedRun\n readonly integrity?: RunIntegrityReport\n constructor(failed: FailedRun, integrity?: RunIntegrityReport) {\n super(`cell ${failed.variantId}/${failed.scenarioId}@${failed.seed} failed: ${failed.reason}`)\n this.failed = failed\n this.integrity = integrity\n }\n}\n\n/**\n * Abort a run whose owning work threw or was orphaned by a sibling's genuine\n * error, finalizing the emitter so hooks fire and the store records a terminal\n * status. Safe to call unconditionally: it only aborts runs that are still\n * `running`. Two no-op cases are intentional and benign:\n *\n * - the run was never started (absent from the store) — nothing to finalize\n * - the run is already terminal (`completed` / `failed` / `aborted`) —\n * finalizing again would overwrite the real outcome (e.g. flip a passed run\n * to `aborted` with `{ pass: false, notes: reason }`), so we leave it alone\n *\n * Any OTHER failure of the store read or the abort write (disk full, FS fault,\n * backend down) is a genuine diagnostic and propagates rather than being\n * swallowed.\n */\nexport async function finalizeAbort(\n emitter: TraceEmitter,\n runId: string,\n reason: string,\n): Promise<void> {\n const existing = await emitter.traceStore.getRun(runId)\n if (existing === undefined) return // run never started; nothing to abort\n if (existing.status !== 'running') return // already finalized; never overwrite a real outcome\n await emitter.abortRun(reason)\n}\n\nfunction defaultRawSinkFactory(workDir: string | undefined) {\n return (params: CampaignFactoryParams): RawProviderSink => {\n if (!workDir) {\n throw new Error(\n 'runEvalCampaign: rawSinkFactory not supplied and workDir not set. Pass either to enable raw provider capture, or pass `new NoopRawProviderSink()` via rawSinkFactory to opt out explicitly.',\n )\n }\n return new FileSystemRawProviderSink({\n dir: `${workDir}/raw-events/${params.runId}`,\n })\n }\n}\n\nasync function resolveAgentProfileCell(\n input: AgentProfileCell | AgentProfileCellInput,\n): Promise<AgentProfileCell> {\n if (isAgentProfileCell(input)) {\n if (!(await verifyAgentProfileCell(input))) {\n throw new Error(`runEvalCampaign: agentProfile.cellId does not match its content`)\n }\n return input\n }\n return buildAgentProfileCell(input)\n}\n\nfunction isAgentProfileCell(\n input: AgentProfileCell | AgentProfileCellInput,\n): input is AgentProfileCell {\n return 'schemaVersion' in input && 'cellId' in input\n}\n\nfunction assertAgentProfileMatchesRun(\n profile: AgentProfileCell,\n model: string,\n promptHash: string,\n): void {\n if (profile.model !== undefined && profile.model !== model) {\n throw new Error(\n `runEvalCampaign: agentProfile.model \"${profile.model}\" does not match outcome.model \"${model}\"`,\n )\n }\n if (profile.promptHash !== undefined && profile.promptHash !== promptHash) {\n throw new Error(\n `runEvalCampaign: agentProfile.promptHash \"${profile.promptHash}\" does not match outcome.promptHash \"${promptHash}\"`,\n )\n }\n}\n\nfunction defaultRunId(params: CampaignFactoryParams): string {\n // Stable across re-runs: fingerprint of (campaignId, variantId, scenarioId, seed).\n // Caller can override via opts.runId for non-deterministic IDs.\n const base = `${params.campaignId}::${params.variantId}::${params.scenarioId}::${params.seed}`\n // TWO 32-bit FNV-1a accumulators with different offsets, concatenated for a\n // 64-bit-wide id. Not crypto-grade; stability and uniqueness are what this\n // needs. Frozen: the value names a persisted cell, and the second\n // accumulator's offset is part of the identity, so no single-accumulator\n // helper can replace it.\n let h1 = 0x811c9dc5\n let h2 = 0x12345678\n for (let i = 0; i < base.length; i++) {\n const c = base.charCodeAt(i)\n h1 = Math.imul(h1 ^ c, 0x01000193) >>> 0\n h2 = Math.imul(h2 ^ c, 0x9e3779b1) >>> 0\n }\n return `run-${h1.toString(16).padStart(8, '0')}${h2.toString(16).padStart(8, '0')}`\n}\n"],"mappings":";;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;AA8SA,MAAM,oBAA8C;CAClD,aAAa;CACb,8BAA8B;CAC9B,gBAAgB;AAClB;AAEA,eAAsB,gBACpB,MAC6B;CAE7B,IAAI,KAAK,SAAS,WAAW,GAC3B,MAAM,IAAI,MAAM,8CAA8C;CAEhE,IAAI,KAAK,UAAU,WAAW,GAC5B,MAAM,IAAI,MAAM,+CAA+C;CAEjE,MAAM,6BAAa,IAAI,IAAY;CACnC,KAAK,MAAM,KAAK,KAAK,UAAU;EAC7B,IAAI,WAAW,IAAI,EAAE,EAAE,GACrB,MAAM,IAAI,MAAM,0CAA0C,EAAE,GAAG,GAAG;EAEpE,WAAW,IAAI,EAAE,EAAE;CACrB;CACA,MAAM,8BAAc,IAAI,IAAY;CACpC,KAAK,MAAM,KAAK,KAAK,WAAW;EAC9B,IAAI,YAAY,IAAI,EAAE,UAAU,GAC9B,MAAM,IAAI,MAAM,0CAA0C,EAAE,WAAW,GAAG;EAE5E,YAAY,IAAI,EAAE,UAAU;CAC9B;CACA,IAAI,KAAK,QAAQ,cAAc,CAAC,WAAW,IAAI,KAAK,OAAO,UAAU,GACnE,MAAM,IAAI,MACR,uCAAuC,KAAK,OAAO,WAAW,iCAChE;CAEF,IAAI,CAAC,KAAK,WACR,MAAM,IAAI,MAAM,oEAAoE;CAGtF,MAAM,QAAQ,KAAK,SAAS;EAAC;EAAG;EAAG;CAAC;CACpC,MAAM,WAAwB,KAAK,YAAY;CAC/C,MAAM,cAAc,KAAK,IAAI,GAAG,KAAK,eAAe,CAAC;CACrD,MAAM,YAAY;EAAE,GAAG;EAAmB,GAAI,KAAK,aAAa,CAAC;CAAG;CACpE,MAAM,qBAA8C,KAAK,sBAAsB;CAC/E,MAAM,MAAM,KAAK,cAAc,KAAK,IAAI;CACxC,MAAM,eAAe,KAAK,gBAAgB;CAC1C,MAAM,sBAAsB,KAAK,uBAAuB;CAExD,MAAM,iBAAiB,KAAK,kBAAkB,sBAAsB,KAAK,OAAO;CAGhF,MAAM,sBAAsB,MAAM,SAAS;EACzC,YAAY,KAAK;EACjB,UAAU,KAAK,SAAS,KAAK,MAAM,EAAE,EAAE,CAAC,CAAC,KAAK;EAC9C,WAAW,KAAK,UAAU,KAAK,MAAM,EAAE,UAAU,CAAC,CAAC,KAAK;EACxD,OAAO,CAAC,GAAG,KAAK,CAAC,CAAC,MAAM,GAAG,MAAM,IAAI,CAAC;EACtC;EACA,YAAY,KAAK,QAAQ,cAAc;EACvC;EACA;CACF,CAAC;CAID,MAAM,QAAgB,CAAC;CACvB,KAAK,MAAM,WAAW,KAAK,UACzB,KAAK,MAAM,YAAY,KAAK,WAC1B,KAAK,MAAM,QAAQ,OACjB,MAAM,KAAK;EAAE;EAAS;EAAU;CAAK,CAAC;CAK5C,MAAM,YAAY,IAAI,KAAK,IAAI,CAAC,CAAC,CAAC,YAAY;CAC9C,MAAM,OAAoB,CAAC;CAC3B,MAAM,mBAAyC,CAAC;CAChD,MAAM,aAA0B,CAAC;CASjC,IAAI,SAAS;CACb,IAAI,WAAW;CACf,MAAM,gBAA2B,CAAC;CAKlC,MAAM,2BAAW,IAAI,IAA0B;CAE/C,eAAe,SAAwB;EACrC,OAAO,CAAC,UAAU;GAChB,MAAM,IAAI;GACV,IAAI,KAAK,MAAM,QAAQ;GACvB,MAAM,OAAO,MAAM;GACnB,IAAI;IACF,MAAM,SAAS,MAAM,WAAW,IAAI;IACpC,KAAK,KAAK,OAAO,MAAM;IACvB,iBAAiB,KAAK,OAAO,SAAS;GACxC,SAAS,KAAK;IACZ,IAAI,eAAe,oBAAoB;KACrC,WAAW,KAAK,IAAI,MAAM;KAC1B,IAAI,IAAI,WAAW,iBAAiB,KAAK,IAAI,SAAS;IACxD,OAAO;KAKL,cAAc,KAAK,GAAG;KACtB,WAAW;KACX;IACF;GACF;EACF;CACF;CAEA,eAAe,WACb,MAC+D;EAC/D,MAAM,SAAS,KAAK,SAAS,aAAA,CAAc;GACzC,YAAY,KAAK;GACjB,OAAO;GACP,WAAW,KAAK,QAAQ;GACxB,YAAY,KAAK,SAAS;GAC1B,MAAM,KAAK;EACb,CAAC;EACD,MAAM,gBAAuC;GAC3C,YAAY,KAAK;GACjB;GACA,WAAW,KAAK,QAAQ;GACxB,YAAY,KAAK,SAAS;GAC1B,MAAM,KAAK;EACb;EACA,MAAM,QAAQ,KAAK,aAAa,aAAa;EAC7C,MAAM,UAAU,eAAe,aAAa;EAE5C,MAAM,UAAU,IAAI,aAAa,OAAO;GACtC;GACA,KAAK,KAAK;GACV,eAAe,KAAK;EACtB,CAAC;EAGD,SAAS,IAAI,OAAO,OAAO;EAE3B,MAAM,OAAO,KAAK,YAAY;GAAE;GAAS;EAAM,CAAC;EAEhD,MAAM,MAA6B;GACjC;GACA,cAAc,KAAK;GACnB,SAAS,KAAK,QAAQ;GACtB,WAAW,KAAK,QAAQ;GACxB,YAAY,KAAK,SAAS;GAC1B,cAAc,KAAK,SAAS,QAAQ,CAAC;GACrC,MAAM,KAAK;GACX;GACA;GACA;GACA;GACA;EACF;EAEA,IAAI;GACF,MAAM,YAAY,IAAI;GACtB,IAAI;GACJ,IAAI;IACF,UAAU,MAAM,KAAK,OAAO,GAAG;GACjC,SAAS,KAAK;IACZ,MAAM,UAAU,eAAe,QAAQ,IAAI,UAAU,OAAO,GAAG;IAM/D,MAAM,cAAc,SAAS,OAAO,OAAO;IAC3C,MAAM,IAAI,mBAAmB;KAC3B;KACA,WAAW,KAAK,QAAQ;KACxB,YAAY,KAAK,SAAS;KAC1B,MAAM,KAAK;KACX,QAAQ;KACR,OAAO;IACT,CAAC;GACH;GACA,MAAM,SAAS,IAAI,IAAI;GAEvB,MAAM,kBAAkB,MAAM,kBAAkB,OAAO,OAAO;IAAE,GAAG;IAAW;GAAQ,CAAC;GACvF,IAAI,CAAC,gBAAgB,IACnB,QAAQ,oBAAR;IACE,KAAK,SACH,MAAM,IAAI,kBAAkB,eAAe;IAC7C,KAAK,eACH,MAAM,IAAI,mBACR;KACE;KACA,WAAW,KAAK,QAAQ;KACxB,YAAY,KAAK,SAAS;KAC1B,MAAM,KAAK;KACX,QAAQ;KACR,OAAO,gBAAgB,OAAO,KAAK,MAAM,EAAE,IAAI,CAAC,CAAC,KAAK,IAAI;IAC5D,GACA,eACF;IACF,KAAK,OAEH;GACJ;GAGF,MAAM,gBAA4B,EAChC,KAAK,QAAQ,OAAO,CAAC,EACvB;GACA,IAAI,aAAa,WAAW,cAAc,eAAe,QAAQ;QAC5D,cAAc,cAAc,QAAQ;GACzC,IAAI,QAAQ,gBAAgB,KAAA,GAAW,cAAc,cAAc,QAAQ;GAE3E,MAAM,SAAoB;IACxB;IACA,cAAc,KAAK;IACnB,aAAa,KAAK,QAAQ;IAC1B,MAAM,KAAK;IACX,OAAO,QAAQ;IACf,YAAY,QAAQ;IACpB,YAAY,QAAQ;IACpB,WAAW,KAAK;IAChB;IACA,SAAS,QAAQ;IACjB,gBAAgB,QAAQ;IACxB,YAAY,QAAQ;IACpB,iBAAiB;IACjB,eAAe,QAAQ;IACvB,SAAS;IACT,GAAI,QAAQ,eAAe,EAAE,cAAc,QAAQ,aAAa,IAAI,CAAC;IACrE,aAAa,QAAQ;IACrB;IACA,YAAY,KAAK,SAAS;GAC5B;GACA,MAAM,gBACJ,QAAQ,iBACP,OAAO,KAAK,iBAAiB,aAC1B,MAAM,KAAK,aAAa;IACtB,YAAY,KAAK;IACjB;IACA,WAAW,KAAK,QAAQ;IACxB,YAAY,KAAK,SAAS;IAC1B,MAAM,KAAK;IACX,SAAS,KAAK,QAAQ;IACtB,cAAc,KAAK,SAAS,QAAQ,CAAC;GACvC,CAAC,IACD,KAAK;GACX,IAAI,kBAAkB,KAAA,GAAW;IAC/B,MAAM,eAAe,MAAM,wBAAwB,aAAa;IAChE,6BAA6B,cAAc,QAAQ,OAAO,QAAQ,UAAU;IAC5E,OAAO,eAAe;GACxB;GACA,OAAO;IAAE,QAAQ,kBAAkB,MAAM;IAAG,WAAW;GAAgB;EACzE,UAAU;GAIR,SAAS,OAAO,KAAK;EACvB;CACF;CAEA,MAAM,UAAU,MAAM,KAAK,EAAE,QAAQ,KAAK,IAAI,aAAa,MAAM,MAAM,EAAE,SAAS,OAAO,CAAC;CAI1F,MAAM,QAAQ,WAAW,OAAO;CAMhC,KAAK,MAAM,CAAC,OAAO,YAAY,UAC7B,MAAM,cAAc,SAAS,OAAO,kDAAkD;CAExF,SAAS,MAAM;CAEf,IAAI,cAAc,SAAS,GACzB,MAAM,cAAc,WAAW,IAC3B,cAAc,KACd,IAAI,eACF,eACA,oBAAoB,cAAc,OAAO,iDAC3C;CAIN,IAAI;CACJ,IAAI,KAAK,QAQP,SAAS,MAAM,eAAe,MAAM;EANlC,GAAG,KAAK;EACR,YAAY,KAAK,OAAO;EACxB,OAAO,aAAa,QAAQ,WAAW;EACvC,aAAa,IAAI,KAAK,IAAI,CAAC,CAAC,CAAC,YAAY;EACzC,qBAAqB,uBAAuB,KAAA;CAED,CAAC;CAGhD,MAAM,UAAU,IAAI,KAAK,IAAI,CAAC,CAAC,CAAC,YAAY;CAE5C,OAAO;EACL,YAAY,KAAK;EACjB;EACA;EACA;EACA;EACA;EACA;EACA;EACA;CACF;AACF;AAIA,IAAM,qBAAN,cAAiC,MAAM;CACrC;CACA;CACA,YAAY,QAAmB,WAAgC;EAC7D,MAAM,QAAQ,OAAO,UAAU,GAAG,OAAO,WAAW,GAAG,OAAO,KAAK,WAAW,OAAO,QAAQ;EAC7F,KAAK,SAAS;EACd,KAAK,YAAY;CACnB;AACF;;;;;;;;;;;;;;;;AAiBA,eAAsB,cACpB,SACA,OACA,QACe;CACf,MAAM,WAAW,MAAM,QAAQ,WAAW,OAAO,KAAK;CACtD,IAAI,aAAa,KAAA,GAAW;CAC5B,IAAI,SAAS,WAAW,WAAW;CACnC,MAAM,QAAQ,SAAS,MAAM;AAC/B;AAEA,SAAS,sBAAsB,SAA6B;CAC1D,QAAQ,WAAmD;EACzD,IAAI,CAAC,SACH,MAAM,IAAI,MACR,6LACF;EAEF,OAAO,IAAI,0BAA0B,EACnC,KAAK,GAAG,QAAQ,cAAc,OAAO,QACvC,CAAC;CACH;AACF;AAEA,eAAe,wBACb,OAC2B;CAC3B,IAAI,mBAAmB,KAAK,GAAG;EAC7B,IAAI,CAAE,MAAM,uBAAuB,KAAK,GACtC,MAAM,IAAI,MAAM,iEAAiE;EAEnF,OAAO;CACT;CACA,OAAO,sBAAsB,KAAK;AACpC;AAEA,SAAS,mBACP,OAC2B;CAC3B,OAAO,mBAAmB,SAAS,YAAY;AACjD;AAEA,SAAS,6BACP,SACA,OACA,YACM;CACN,IAAI,QAAQ,UAAU,KAAA,KAAa,QAAQ,UAAU,OACnD,MAAM,IAAI,MACR,wCAAwC,QAAQ,MAAM,kCAAkC,MAAM,EAChG;CAEF,IAAI,QAAQ,eAAe,KAAA,KAAa,QAAQ,eAAe,YAC7D,MAAM,IAAI,MACR,6CAA6C,QAAQ,WAAW,uCAAuC,WAAW,EACpH;AAEJ;AAEA,SAAS,aAAa,QAAuC;CAG3D,MAAM,OAAO,GAAG,OAAO,WAAW,IAAI,OAAO,UAAU,IAAI,OAAO,WAAW,IAAI,OAAO;CAMxF,IAAI,KAAK;CACT,IAAI,KAAK;CACT,KAAK,IAAI,IAAI,GAAG,IAAI,KAAK,QAAQ,KAAK;EACpC,MAAM,IAAI,KAAK,WAAW,CAAC;EAC3B,KAAK,KAAK,KAAK,KAAK,GAAG,QAAU,MAAM;EACvC,KAAK,KAAK,KAAK,KAAK,GAAG,UAAU,MAAM;CACzC;CACA,OAAO,OAAO,GAAG,SAAS,EAAE,CAAC,CAAC,SAAS,GAAG,GAAG,IAAI,GAAG,SAAS,EAAE,CAAC,CAAC,SAAS,GAAG,GAAG;AAClF"}
|
|
1
|
+
{"version":3,"file":"eval-campaign-Cs-7MiCs.js","names":[],"sources":["../src/eval-campaign.ts"],"sourcesContent":["/**\n * EvalCampaign — opinionated matrix runner that wires the four\n * capture-integrity directives by construction.\n *\n * The canonical benchmark shape — matrix runner → for each\n * (variant, scenario, seed) → start a TraceEmitter → call LLMs → end the\n * run → analyze — has a bug class at the integration boundary: raw\n * events not captured, route silently wrong, integrity not asserted,\n * analyst never run. The directives in `SKILL.md § Capture integrity`\n * are the mitigations.\n *\n * `EvalCampaign` is the structural fix — consumers don't wire the\n * integrity surface themselves; the campaign owns it. Specifically:\n *\n * - constructs a per-run `TraceStore` and `RawProviderSink` via factories\n * - builds the run's `ChatClient` through `chatFactory`, handing it the\n * run's raw sink and trace context — a transport built any other way has\n * no raw HTTP envelope, and `assertRunCaptured` says so\n * - calls `assertRunCaptured` after every `endRun` and routes failures\n * through a configurable policy (`throw` / `mark_failed` / `log`)\n * - assembles per-run `RunRecord`s and runs `researchReport` at the end\n * so the campaign artifact is launch-decision-grade by default\n * - embeds the campaign fingerprint (a SHA-256 over the canonicalised\n * run set) and optional `preregistrationHash` in the report\n *\n * The runner contract is intentionally narrow: produce a `CampaignRunOutcome`\n * given a fully-wired `CampaignRunContext`. Everything orchestration-shaped\n * lives in the campaign. This is the inversion-of-control point — consumers\n * stop writing matrix runners and start writing scenario-runners.\n *\n * Out of scope for v1 (tracked in `docs/research-report-methodology.md`):\n *\n * - Distributed/cluster execution (concurrency is local async)\n * - Adaptive sampling / sequential interim looks\n * - Resume from partial state across crashes\n * - LLM-call retry beyond what the caller's transport already does\n */\n\nimport {\n type AgentProfileCell,\n type AgentProfileCellInput,\n buildAgentProfileCell,\n verifyAgentProfileCell,\n} from './agent-profile-cell'\nimport type { ChatClient } from './analyst/chat-client'\nimport { hashJson } from './pre-registration'\nimport type {\n JudgeScoresRecord,\n RunCostProvenance,\n RunJudgeMetadata,\n RunOutcome,\n RunRecord,\n RunSplitTag,\n RunTaskFailure,\n RunTokenUsage,\n} from './run-record'\nimport { validateRunRecord } from './run-record'\nimport { type ResearchReport, type ResearchReportOptions, researchReport } from './summary-report'\nimport type { RunCompleteHook } from './trace/emitter'\nimport { TraceEmitter } from './trace/emitter'\nimport {\n assertRunCaptured,\n RunIntegrityError,\n type RunIntegrityExpectations,\n type RunIntegrityReport,\n} from './trace/integrity'\nimport { FileSystemRawProviderSink, type RawProviderSink } from './trace/raw-provider-sink'\nimport type { TraceStore } from './trace/store'\n\n// ── Public types ─────────────────────────────────────────────────────────\n\nexport interface CampaignVariant<V> {\n id: string\n payload: V\n}\n\nexport interface CampaignScenario {\n scenarioId: string\n /** Free-form metadata propagated to runs and reports. */\n tags?: Record<string, string>\n}\n\nexport interface CampaignRunContext<V> {\n /** Stable run id. The campaign generates this; the runner does not. */\n runId: string\n /** Logical experiment id (campaignId by default; overridable per-run via opts). */\n experimentId: string\n variant: V\n variantId: string\n scenarioId: string\n scenarioTags: Record<string, string>\n seed: number\n splitTag: RunSplitTag\n /**\n * The TraceEmitter for this run, with `onRunComplete` hooks pre-wired\n * (analyst auto-execution if configured, plus integrity check). The\n * runner MUST call `emitter.startRun` before doing any work and either\n * `emitter.endRun` or `emitter.abortRun` before returning.\n */\n emitter: TraceEmitter\n store: TraceStore\n rawSink: RawProviderSink\n /**\n * The run's model transport, built by `chatFactory` with this run's\n * `rawSink` and `runId` already bound.\n */\n chat: ChatClient\n}\n\n/** What the campaign binds into the run's transport. */\nexport interface CampaignChatWiring {\n /**\n * Raw provider sink for this run. Bind it into the transport: the campaign's\n * integrity check requires every LLM span to carry a matching raw request\n * event, so a transport built without it fails `assertRunCaptured`.\n */\n rawSink: RawProviderSink\n runId: string\n}\n\ninterface CampaignRunOutcomeFields {\n /** Did the run pass? Mirrors `RunOutcome.pass` semantics. */\n pass: boolean\n /** Score for the run on its split. Maps to `searchScore` or `holdoutScore`. */\n score: number\n /** Cost in USD, or null when the runner could not capture it. */\n costUsd: number | null\n /** Source of the cost amount. */\n costProvenance: RunCostProvenance\n tokenUsage: RunTokenUsage\n /** Snapshot model id (e.g. `claude-sonnet-4-6@2025-04-15`). */\n model: string\n /** sha256 of the effective prompt sent to the model. */\n promptHash: string\n /** sha256 of the effective config (model, temperature, tools, judges, splits). */\n configHash: string\n /** Optional extra numeric metrics to land in `outcome.raw`. */\n raw?: Record<string, number>\n /** Optional judge metadata when a judge was used. */\n judgeMetadata?: RunJudgeMetadata\n /**\n * Optional per-judge / per-dim breakdown for ensemble-judged runs.\n * Propagated to `outcome.judgeScores` on the resulting `RunRecord`.\n * Single-judge or scalar-only runs leave this unset.\n */\n judgeScores?: JudgeScoresRecord\n /**\n * Agent profile cell observed by the runner. When supplied, it overrides\n * `EvalCampaignOptions.agentProfile` for this run and must match the\n * outcome's `model` and `promptHash`.\n */\n agentProfile?: AgentProfileCell | AgentProfileCellInput\n}\n\n/** Campaign result with the same task-failure invariant as `RunRecord`. */\nexport type CampaignRunOutcome = CampaignRunOutcomeFields & RunTaskFailure\n\nexport type CampaignRunner<V> = (ctx: CampaignRunContext<V>) => Promise<CampaignRunOutcome>\n\nexport type CampaignIntegrityPolicy = 'throw' | 'mark_failed' | 'log'\n\nexport interface EvalCampaignOptions<V> {\n /**\n * Stable id for the campaign. Used as the default `experimentId` on\n * every run, and folded into the campaign fingerprint.\n */\n campaignId: string\n variants: CampaignVariant<V>[]\n scenarios: CampaignScenario[]\n /** Default `[0, 1, 2]`. */\n seeds?: number[]\n /** Default `'holdout'` — the split that anchors a launch decision. */\n splitTag?: RunSplitTag\n /** Git SHA the campaign is run against. Mandatory; `RunRecord` rejects unset. */\n commitSha: string\n /**\n * Build the model transport for one run. agent-eval executes no paid model:\n * the caller owns the transport and the credential never enters this\n * package. The campaign calls this once per run and passes the run's raw\n * provider sink and `runId`, so a transport that binds them captures the\n * raw HTTP envelope `assertRunCaptured` checks for.\n */\n chatFactory: (wiring: CampaignChatWiring) => ChatClient\n /**\n * Caller-declared identity of the execution route, folded into the campaign\n * fingerprint so two campaigns run against different endpoints do not share\n * one identity. agent-eval no longer knows the endpoint; the owner of\n * execution names it.\n */\n executionRef?: string\n /**\n * Per-run TraceStore factory. Common shape: a fresh store per run keyed\n * on `runId`. Implementations that share a store across the campaign\n * are valid — the campaign only writes through `emitter`.\n */\n storeFactory: (params: CampaignFactoryParams) => TraceStore\n /**\n * Per-run RawProviderSink factory. Defaults to `FileSystemRawProviderSink`\n * rooted at `${workDir}/raw-events/${runId}` if `workDir` is supplied;\n * otherwise required. Forensic capture is non-negotiable in a campaign\n * run — pass `NoopRawProviderSink` explicitly if you want to opt out.\n */\n rawSinkFactory?: (params: CampaignFactoryParams) => RawProviderSink\n /**\n * Filesystem root for default `rawSinkFactory`. Ignored if\n * `rawSinkFactory` is supplied.\n */\n workDir?: string\n /**\n * Extra `onRunComplete` hooks the campaign appends (after its own\n * integrity-check hook). Pass `traceAnalystOnRunComplete(...)` here.\n */\n onRunComplete?: RunCompleteHook[]\n /**\n * Per-run integrity expectations. Defaults to:\n * `{ llmSpansMin: 1, requireRawCoverageOfLlmSpans: true, requireOutcome: true }`.\n * Override (e.g. `{ llmSpansMin: 0 }`) for runs that don't call LLMs.\n */\n integrity?: RunIntegrityExpectations\n /** Behaviour when integrity fails. Default `'mark_failed'`. */\n onIntegrityFailure?: CampaignIntegrityPolicy\n /**\n * Per-run runner. Receives a fully-wired context; produces an outcome\n * the campaign converts into a `RunRecord`.\n */\n runner: CampaignRunner<V>\n /**\n * If set, the campaign computes `researchReport` at the end. `comparator`\n * is a `variantId`. Other fields are forwarded verbatim.\n */\n report?: { comparator?: string } & Omit<\n ResearchReportOptions,\n 'comparator' | 'preregistrationHash' | 'generatedAt'\n >\n /**\n * Hash of a signed `HypothesisManifest` (see `pre-registration.ts`).\n * Embedded in the campaign fingerprint and the research report.\n */\n preregistrationHash?: string\n /** Local concurrency. Default `1` (sequential). */\n concurrency?: number\n /**\n * Override the time source. Tests pass a mock to make wallMs deterministic.\n */\n now?: () => number\n /** Override the runId generator. Tests pin this. */\n runId?: (params: CampaignFactoryParams) => string\n /**\n * Agent profile cell for campaign runs. Static profiles can pass an object;\n * routers or variant-specific harnesses can pass a factory. The campaign\n * stamps the built cell onto every `RunRecord` and rejects profile/model or\n * profile/prompt contradictions.\n */\n agentProfile?:\n | AgentProfileCell\n | AgentProfileCellInput\n | ((\n params: CampaignFactoryParams & {\n variant: V\n scenarioTags: Record<string, string>\n },\n ) =>\n | AgentProfileCell\n | AgentProfileCellInput\n | Promise<AgentProfileCell | AgentProfileCellInput>)\n}\n\nexport interface CampaignFactoryParams {\n campaignId: string\n runId: string\n variantId: string\n scenarioId: string\n seed: number\n}\n\nexport interface FailedRun {\n runId: string\n variantId: string\n scenarioId: string\n seed: number\n reason: string\n error?: string\n}\n\nexport interface EvalCampaignResult {\n campaignId: string\n /** SHA-256 over canonicalised `(variantIds, scenarioIds, seeds, comparator, splitTag, executionRef, preregistrationHash)`. */\n campaignFingerprint: string\n preregistrationHash: string | null\n /** Successful runs only. Failed runs land in `failedRuns`. */\n runs: RunRecord[]\n /** Integrity reports for every successful run. */\n integrityReports: RunIntegrityReport[]\n failedRuns: FailedRun[]\n /** Computed when `report` is set on options. */\n report?: ResearchReport\n startedAt: string\n endedAt: string\n}\n\n// ── Implementation ───────────────────────────────────────────────────────\n\nconst DEFAULT_INTEGRITY: RunIntegrityExpectations = {\n llmSpansMin: 1,\n requireRawCoverageOfLlmSpans: true,\n requireOutcome: true,\n}\n\nexport async function runEvalCampaign<V>(\n opts: EvalCampaignOptions<V>,\n): Promise<EvalCampaignResult> {\n // ── Preflight ──────────────────────────────────────────────────────\n if (opts.variants.length === 0) {\n throw new Error('runEvalCampaign: variants must be non-empty.')\n }\n if (opts.scenarios.length === 0) {\n throw new Error('runEvalCampaign: scenarios must be non-empty.')\n }\n const variantIds = new Set<string>()\n for (const v of opts.variants) {\n if (variantIds.has(v.id)) {\n throw new Error(`runEvalCampaign: duplicate variant id \"${v.id}\".`)\n }\n variantIds.add(v.id)\n }\n const scenarioIds = new Set<string>()\n for (const s of opts.scenarios) {\n if (scenarioIds.has(s.scenarioId)) {\n throw new Error(`runEvalCampaign: duplicate scenarioId \"${s.scenarioId}\".`)\n }\n scenarioIds.add(s.scenarioId)\n }\n if (opts.report?.comparator && !variantIds.has(opts.report.comparator)) {\n throw new Error(\n `runEvalCampaign: report.comparator \"${opts.report.comparator}\" is not a configured variantId.`,\n )\n }\n if (!opts.commitSha) {\n throw new Error('runEvalCampaign: commitSha is required (every RunRecord needs it).')\n }\n\n const seeds = opts.seeds ?? [0, 1, 2]\n const splitTag: RunSplitTag = opts.splitTag ?? 'holdout'\n const concurrency = Math.max(1, opts.concurrency ?? 1)\n const integrity = { ...DEFAULT_INTEGRITY, ...(opts.integrity ?? {}) }\n const onIntegrityFailure: CampaignIntegrityPolicy = opts.onIntegrityFailure ?? 'mark_failed'\n const now = opts.now ?? (() => Date.now())\n const executionRef = opts.executionRef ?? null\n const preregistrationHash = opts.preregistrationHash ?? null\n\n const rawSinkFactory = opts.rawSinkFactory ?? defaultRawSinkFactory(opts.workDir)\n\n // ── Fingerprint ────────────────────────────────────────────────────\n const campaignFingerprint = await hashJson({\n campaignId: opts.campaignId,\n variants: opts.variants.map((v) => v.id).sort(),\n scenarios: opts.scenarios.map((s) => s.scenarioId).sort(),\n seeds: [...seeds].sort((a, b) => a - b),\n splitTag,\n comparator: opts.report?.comparator ?? null,\n executionRef,\n preregistrationHash,\n })\n\n // ── Plan the matrix ────────────────────────────────────────────────\n type Cell = { variant: CampaignVariant<V>; scenario: CampaignScenario; seed: number }\n const cells: Cell[] = []\n for (const variant of opts.variants) {\n for (const scenario of opts.scenarios) {\n for (const seed of seeds) {\n cells.push({ variant, scenario, seed })\n }\n }\n }\n\n const startedAt = new Date(now()).toISOString()\n const runs: RunRecord[] = []\n const integrityReports: RunIntegrityReport[] = []\n const failedRuns: FailedRun[] = []\n\n // ── Execute (bounded-concurrency worker pool) ──────────────────────\n // A genuine (non-CellExecutionError) error from any worker is a bug, not a\n // run-level failure. We must NOT let `Promise.all` reject mid-flight and\n // orphan the other workers' in-progress runs (emitters never finalized,\n // sinks/handles leak, partial work silently discarded). Instead: capture the\n // first genuine error, stop dispatching new cells so in-flight workers wind\n // down, finalize every still-open run, then re-throw the aggregated error.\n let cursor = 0\n let aborting = false\n const genuineErrors: unknown[] = []\n // Emitters for runs that are currently mid-flight, keyed by runId. A worker\n // registers its emitter before invoking the runner and removes it once the\n // run is finalized (via endRun/abortRun, success or failure). Anything left\n // here after the pool settles is an orphan we must abort.\n const openRuns = new Map<string, TraceEmitter>()\n\n async function worker(): Promise<void> {\n while (!aborting) {\n const i = cursor++\n if (i >= cells.length) return\n const cell = cells[i]!\n try {\n const result = await runOneCell(cell)\n runs.push(result.record)\n integrityReports.push(result.integrity)\n } catch (err) {\n if (err instanceof CellExecutionError) {\n failedRuns.push(err.failed)\n if (err.integrity) integrityReports.push(err.integrity)\n } else {\n // Genuine bug — not a runner failure, not an integrity failure.\n // Capture it and stop dispatching so peers wind down gracefully;\n // re-thrown after the pool settles. Do not surface here — that would\n // reject Promise.all and orphan the other workers' runs.\n genuineErrors.push(err)\n aborting = true\n return\n }\n }\n }\n }\n\n async function runOneCell(\n cell: Cell,\n ): Promise<{ record: RunRecord; integrity: RunIntegrityReport }> {\n const runId = (opts.runId ?? defaultRunId)({\n campaignId: opts.campaignId,\n runId: '', // unused by default generator\n variantId: cell.variant.id,\n scenarioId: cell.scenario.scenarioId,\n seed: cell.seed,\n })\n const factoryParams: CampaignFactoryParams = {\n campaignId: opts.campaignId,\n runId,\n variantId: cell.variant.id,\n scenarioId: cell.scenario.scenarioId,\n seed: cell.seed,\n }\n const store = opts.storeFactory(factoryParams)\n const rawSink = rawSinkFactory(factoryParams)\n\n const emitter = new TraceEmitter(store, {\n runId,\n now: opts.now,\n onRunComplete: opts.onRunComplete,\n })\n // Track this run as open so a genuine error elsewhere in the pool can\n // finalize it instead of orphaning it. Removed in the finally below.\n openRuns.set(runId, emitter)\n\n const chat = opts.chatFactory({ rawSink, runId })\n\n const ctx: CampaignRunContext<V> = {\n runId,\n experimentId: opts.campaignId,\n variant: cell.variant.payload,\n variantId: cell.variant.id,\n scenarioId: cell.scenario.scenarioId,\n scenarioTags: cell.scenario.tags ?? {},\n seed: cell.seed,\n splitTag,\n emitter,\n store,\n rawSink,\n chat,\n }\n\n try {\n const wallStart = now()\n let outcome: CampaignRunOutcome\n try {\n outcome = await opts.runner(ctx)\n } catch (err) {\n const message = err instanceof Error ? err.message : String(err)\n // The runner threw mid-execution. Abort the run so the emitter\n // finalizes. The only benign abortRun failure is \"the runner never\n // started the run\" (nothing to finalize). A store-write failure (disk\n // full, FS error) is a genuine diagnostic — surface it rather than\n // masking it as a plain runner failure.\n await finalizeAbort(emitter, runId, message)\n throw new CellExecutionError({\n runId,\n variantId: cell.variant.id,\n scenarioId: cell.scenario.scenarioId,\n seed: cell.seed,\n reason: 'runner_threw',\n error: message,\n })\n }\n const wallMs = now() - wallStart\n\n const integrityReport = await assertRunCaptured(store, runId, { ...integrity, rawSink })\n if (!integrityReport.ok) {\n switch (onIntegrityFailure) {\n case 'throw':\n throw new RunIntegrityError(integrityReport)\n case 'mark_failed':\n throw new CellExecutionError(\n {\n runId,\n variantId: cell.variant.id,\n scenarioId: cell.scenario.scenarioId,\n seed: cell.seed,\n reason: 'integrity_failed',\n error: integrityReport.issues.map((i) => i.code).join(', '),\n },\n integrityReport,\n )\n case 'log':\n // Caller wants the run admitted with a flagged report; fall through.\n break\n }\n }\n\n const recordOutcome: RunOutcome = {\n raw: outcome.raw ?? {},\n }\n if (splitTag === 'holdout') recordOutcome.holdoutScore = outcome.score\n else recordOutcome.searchScore = outcome.score\n if (outcome.judgeScores !== undefined) recordOutcome.judgeScores = outcome.judgeScores\n\n const record: RunRecord = {\n runId,\n experimentId: opts.campaignId,\n candidateId: cell.variant.id,\n seed: cell.seed,\n model: outcome.model,\n promptHash: outcome.promptHash,\n configHash: outcome.configHash,\n commitSha: opts.commitSha,\n wallMs,\n costUsd: outcome.costUsd,\n costProvenance: outcome.costProvenance,\n tokenUsage: outcome.tokenUsage,\n terminalOutcome: 'succeeded',\n judgeMetadata: outcome.judgeMetadata,\n outcome: recordOutcome,\n ...(outcome.failureClass ? { failureClass: outcome.failureClass } : {}),\n failureMode: outcome.failureMode,\n splitTag,\n scenarioId: cell.scenario.scenarioId,\n }\n const profileSource =\n outcome.agentProfile ??\n (typeof opts.agentProfile === 'function'\n ? await opts.agentProfile({\n campaignId: opts.campaignId,\n runId,\n variantId: cell.variant.id,\n scenarioId: cell.scenario.scenarioId,\n seed: cell.seed,\n variant: cell.variant.payload,\n scenarioTags: cell.scenario.tags ?? {},\n })\n : opts.agentProfile)\n if (profileSource !== undefined) {\n const agentProfile = await resolveAgentProfileCell(profileSource)\n assertAgentProfileMatchesRun(agentProfile, outcome.model, outcome.promptHash)\n record.agentProfile = agentProfile\n }\n return { record: validateRunRecord(record), integrity: integrityReport }\n } finally {\n // This run's worker has finished with it (success, run-level failure, or\n // genuine error). It is no longer the pool's job to finalize — drop it so\n // the post-settle sweep doesn't double-abort a finalized run.\n openRuns.delete(runId)\n }\n }\n\n const workers = Array.from({ length: Math.min(concurrency, cells.length) }, () => worker())\n // allSettled (not all): a genuine error in one worker must not reject the\n // pool mid-flight and orphan the others. Each worker captures its own\n // genuine error into `genuineErrors` and returns; we re-throw below.\n await Promise.allSettled(workers)\n\n // Finalize any run still open after the pool wound down. With the\n // stop-dispatch flag these are the runs that were mid-flight in peer workers\n // when the first genuine error fired — abort them so their emitters finalize\n // (hooks fire, store records a terminal status) instead of leaking.\n for (const [runId, emitter] of openRuns) {\n await finalizeAbort(emitter, runId, 'campaign aborted: genuine error in a sibling run')\n }\n openRuns.clear()\n\n if (genuineErrors.length > 0) {\n throw genuineErrors.length === 1\n ? genuineErrors[0]\n : new AggregateError(\n genuineErrors,\n `runEvalCampaign: ${genuineErrors.length} runs failed with genuine (non-run-level) errors`,\n )\n }\n\n // ── Optional research report ───────────────────────────────────────\n let report: ResearchReport | undefined\n if (opts.report) {\n const reportOpts: ResearchReportOptions = {\n ...opts.report,\n comparator: opts.report.comparator,\n split: splitTag === 'dev' ? 'search' : splitTag,\n generatedAt: new Date(now()).toISOString(),\n preregistrationHash: preregistrationHash ?? undefined,\n }\n report = await researchReport(runs, reportOpts)\n }\n\n const endedAt = new Date(now()).toISOString()\n\n return {\n campaignId: opts.campaignId,\n campaignFingerprint,\n preregistrationHash,\n runs,\n integrityReports,\n failedRuns,\n report,\n startedAt,\n endedAt,\n }\n}\n\n// ── Internal ─────────────────────────────────────────────────────────────\n\nclass CellExecutionError extends Error {\n readonly failed: FailedRun\n readonly integrity?: RunIntegrityReport\n constructor(failed: FailedRun, integrity?: RunIntegrityReport) {\n super(`cell ${failed.variantId}/${failed.scenarioId}@${failed.seed} failed: ${failed.reason}`)\n this.failed = failed\n this.integrity = integrity\n }\n}\n\n/**\n * Abort a run whose owning work threw or was orphaned by a sibling's genuine\n * error, finalizing the emitter so hooks fire and the store records a terminal\n * status. Safe to call unconditionally: it only aborts runs that are still\n * `running`. Two no-op cases are intentional and benign:\n *\n * - the run was never started (absent from the store) — nothing to finalize\n * - the run is already terminal (`completed` / `failed` / `aborted`) —\n * finalizing again would overwrite the real outcome (e.g. flip a passed run\n * to `aborted` with `{ pass: false, notes: reason }`), so we leave it alone\n *\n * Any OTHER failure of the store read or the abort write (disk full, FS fault,\n * backend down) is a genuine diagnostic and propagates rather than being\n * swallowed.\n */\nexport async function finalizeAbort(\n emitter: TraceEmitter,\n runId: string,\n reason: string,\n): Promise<void> {\n const existing = await emitter.traceStore.getRun(runId)\n if (existing === undefined) return // run never started; nothing to abort\n if (existing.status !== 'running') return // already finalized; never overwrite a real outcome\n await emitter.abortRun(reason)\n}\n\nfunction defaultRawSinkFactory(workDir: string | undefined) {\n return (params: CampaignFactoryParams): RawProviderSink => {\n if (!workDir) {\n throw new Error(\n 'runEvalCampaign: rawSinkFactory not supplied and workDir not set. Pass either to enable raw provider capture, or pass `new NoopRawProviderSink()` via rawSinkFactory to opt out explicitly.',\n )\n }\n return new FileSystemRawProviderSink({\n dir: `${workDir}/raw-events/${params.runId}`,\n })\n }\n}\n\nasync function resolveAgentProfileCell(\n input: AgentProfileCell | AgentProfileCellInput,\n): Promise<AgentProfileCell> {\n if (isAgentProfileCell(input)) {\n if (!(await verifyAgentProfileCell(input))) {\n throw new Error(`runEvalCampaign: agentProfile.cellId does not match its content`)\n }\n return input\n }\n return buildAgentProfileCell(input)\n}\n\nfunction isAgentProfileCell(\n input: AgentProfileCell | AgentProfileCellInput,\n): input is AgentProfileCell {\n return 'schemaVersion' in input && 'cellId' in input\n}\n\nfunction assertAgentProfileMatchesRun(\n profile: AgentProfileCell,\n model: string,\n promptHash: string,\n): void {\n if (profile.model !== undefined && profile.model !== model) {\n throw new Error(\n `runEvalCampaign: agentProfile.model \"${profile.model}\" does not match outcome.model \"${model}\"`,\n )\n }\n if (profile.promptHash !== undefined && profile.promptHash !== promptHash) {\n throw new Error(\n `runEvalCampaign: agentProfile.promptHash \"${profile.promptHash}\" does not match outcome.promptHash \"${promptHash}\"`,\n )\n }\n}\n\nfunction defaultRunId(params: CampaignFactoryParams): string {\n // Stable across re-runs: fingerprint of (campaignId, variantId, scenarioId, seed).\n // Caller can override via opts.runId for non-deterministic IDs.\n const base = `${params.campaignId}::${params.variantId}::${params.scenarioId}::${params.seed}`\n // TWO 32-bit FNV-1a accumulators with different offsets, concatenated for a\n // 64-bit-wide id. Not crypto-grade; stability and uniqueness are what this\n // needs. Frozen: the value names a persisted cell, and the second\n // accumulator's offset is part of the identity, so no single-accumulator\n // helper can replace it.\n let h1 = 0x811c9dc5\n let h2 = 0x12345678\n for (let i = 0; i < base.length; i++) {\n const c = base.charCodeAt(i)\n h1 = Math.imul(h1 ^ c, 0x01000193) >>> 0\n h2 = Math.imul(h2 ^ c, 0x9e3779b1) >>> 0\n }\n return `run-${h1.toString(16).padStart(8, '0')}${h2.toString(16).padStart(8, '0')}`\n}\n"],"mappings":";;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;AA8SA,MAAM,oBAA8C;CAClD,aAAa;CACb,8BAA8B;CAC9B,gBAAgB;AAClB;AAEA,eAAsB,gBACpB,MAC6B;CAE7B,IAAI,KAAK,SAAS,WAAW,GAC3B,MAAM,IAAI,MAAM,8CAA8C;CAEhE,IAAI,KAAK,UAAU,WAAW,GAC5B,MAAM,IAAI,MAAM,+CAA+C;CAEjE,MAAM,6BAAa,IAAI,IAAY;CACnC,KAAK,MAAM,KAAK,KAAK,UAAU;EAC7B,IAAI,WAAW,IAAI,EAAE,EAAE,GACrB,MAAM,IAAI,MAAM,0CAA0C,EAAE,GAAG,GAAG;EAEpE,WAAW,IAAI,EAAE,EAAE;CACrB;CACA,MAAM,8BAAc,IAAI,IAAY;CACpC,KAAK,MAAM,KAAK,KAAK,WAAW;EAC9B,IAAI,YAAY,IAAI,EAAE,UAAU,GAC9B,MAAM,IAAI,MAAM,0CAA0C,EAAE,WAAW,GAAG;EAE5E,YAAY,IAAI,EAAE,UAAU;CAC9B;CACA,IAAI,KAAK,QAAQ,cAAc,CAAC,WAAW,IAAI,KAAK,OAAO,UAAU,GACnE,MAAM,IAAI,MACR,uCAAuC,KAAK,OAAO,WAAW,iCAChE;CAEF,IAAI,CAAC,KAAK,WACR,MAAM,IAAI,MAAM,oEAAoE;CAGtF,MAAM,QAAQ,KAAK,SAAS;EAAC;EAAG;EAAG;CAAC;CACpC,MAAM,WAAwB,KAAK,YAAY;CAC/C,MAAM,cAAc,KAAK,IAAI,GAAG,KAAK,eAAe,CAAC;CACrD,MAAM,YAAY;EAAE,GAAG;EAAmB,GAAI,KAAK,aAAa,CAAC;CAAG;CACpE,MAAM,qBAA8C,KAAK,sBAAsB;CAC/E,MAAM,MAAM,KAAK,cAAc,KAAK,IAAI;CACxC,MAAM,eAAe,KAAK,gBAAgB;CAC1C,MAAM,sBAAsB,KAAK,uBAAuB;CAExD,MAAM,iBAAiB,KAAK,kBAAkB,sBAAsB,KAAK,OAAO;CAGhF,MAAM,sBAAsB,MAAM,SAAS;EACzC,YAAY,KAAK;EACjB,UAAU,KAAK,SAAS,KAAK,MAAM,EAAE,EAAE,CAAC,CAAC,KAAK;EAC9C,WAAW,KAAK,UAAU,KAAK,MAAM,EAAE,UAAU,CAAC,CAAC,KAAK;EACxD,OAAO,CAAC,GAAG,KAAK,CAAC,CAAC,MAAM,GAAG,MAAM,IAAI,CAAC;EACtC;EACA,YAAY,KAAK,QAAQ,cAAc;EACvC;EACA;CACF,CAAC;CAID,MAAM,QAAgB,CAAC;CACvB,KAAK,MAAM,WAAW,KAAK,UACzB,KAAK,MAAM,YAAY,KAAK,WAC1B,KAAK,MAAM,QAAQ,OACjB,MAAM,KAAK;EAAE;EAAS;EAAU;CAAK,CAAC;CAK5C,MAAM,YAAY,IAAI,KAAK,IAAI,CAAC,CAAC,CAAC,YAAY;CAC9C,MAAM,OAAoB,CAAC;CAC3B,MAAM,mBAAyC,CAAC;CAChD,MAAM,aAA0B,CAAC;CASjC,IAAI,SAAS;CACb,IAAI,WAAW;CACf,MAAM,gBAA2B,CAAC;CAKlC,MAAM,2BAAW,IAAI,IAA0B;CAE/C,eAAe,SAAwB;EACrC,OAAO,CAAC,UAAU;GAChB,MAAM,IAAI;GACV,IAAI,KAAK,MAAM,QAAQ;GACvB,MAAM,OAAO,MAAM;GACnB,IAAI;IACF,MAAM,SAAS,MAAM,WAAW,IAAI;IACpC,KAAK,KAAK,OAAO,MAAM;IACvB,iBAAiB,KAAK,OAAO,SAAS;GACxC,SAAS,KAAK;IACZ,IAAI,eAAe,oBAAoB;KACrC,WAAW,KAAK,IAAI,MAAM;KAC1B,IAAI,IAAI,WAAW,iBAAiB,KAAK,IAAI,SAAS;IACxD,OAAO;KAKL,cAAc,KAAK,GAAG;KACtB,WAAW;KACX;IACF;GACF;EACF;CACF;CAEA,eAAe,WACb,MAC+D;EAC/D,MAAM,SAAS,KAAK,SAAS,aAAA,CAAc;GACzC,YAAY,KAAK;GACjB,OAAO;GACP,WAAW,KAAK,QAAQ;GACxB,YAAY,KAAK,SAAS;GAC1B,MAAM,KAAK;EACb,CAAC;EACD,MAAM,gBAAuC;GAC3C,YAAY,KAAK;GACjB;GACA,WAAW,KAAK,QAAQ;GACxB,YAAY,KAAK,SAAS;GAC1B,MAAM,KAAK;EACb;EACA,MAAM,QAAQ,KAAK,aAAa,aAAa;EAC7C,MAAM,UAAU,eAAe,aAAa;EAE5C,MAAM,UAAU,IAAI,aAAa,OAAO;GACtC;GACA,KAAK,KAAK;GACV,eAAe,KAAK;EACtB,CAAC;EAGD,SAAS,IAAI,OAAO,OAAO;EAE3B,MAAM,OAAO,KAAK,YAAY;GAAE;GAAS;EAAM,CAAC;EAEhD,MAAM,MAA6B;GACjC;GACA,cAAc,KAAK;GACnB,SAAS,KAAK,QAAQ;GACtB,WAAW,KAAK,QAAQ;GACxB,YAAY,KAAK,SAAS;GAC1B,cAAc,KAAK,SAAS,QAAQ,CAAC;GACrC,MAAM,KAAK;GACX;GACA;GACA;GACA;GACA;EACF;EAEA,IAAI;GACF,MAAM,YAAY,IAAI;GACtB,IAAI;GACJ,IAAI;IACF,UAAU,MAAM,KAAK,OAAO,GAAG;GACjC,SAAS,KAAK;IACZ,MAAM,UAAU,eAAe,QAAQ,IAAI,UAAU,OAAO,GAAG;IAM/D,MAAM,cAAc,SAAS,OAAO,OAAO;IAC3C,MAAM,IAAI,mBAAmB;KAC3B;KACA,WAAW,KAAK,QAAQ;KACxB,YAAY,KAAK,SAAS;KAC1B,MAAM,KAAK;KACX,QAAQ;KACR,OAAO;IACT,CAAC;GACH;GACA,MAAM,SAAS,IAAI,IAAI;GAEvB,MAAM,kBAAkB,MAAM,kBAAkB,OAAO,OAAO;IAAE,GAAG;IAAW;GAAQ,CAAC;GACvF,IAAI,CAAC,gBAAgB,IACnB,QAAQ,oBAAR;IACE,KAAK,SACH,MAAM,IAAI,kBAAkB,eAAe;IAC7C,KAAK,eACH,MAAM,IAAI,mBACR;KACE;KACA,WAAW,KAAK,QAAQ;KACxB,YAAY,KAAK,SAAS;KAC1B,MAAM,KAAK;KACX,QAAQ;KACR,OAAO,gBAAgB,OAAO,KAAK,MAAM,EAAE,IAAI,CAAC,CAAC,KAAK,IAAI;IAC5D,GACA,eACF;IACF,KAAK,OAEH;GACJ;GAGF,MAAM,gBAA4B,EAChC,KAAK,QAAQ,OAAO,CAAC,EACvB;GACA,IAAI,aAAa,WAAW,cAAc,eAAe,QAAQ;QAC5D,cAAc,cAAc,QAAQ;GACzC,IAAI,QAAQ,gBAAgB,KAAA,GAAW,cAAc,cAAc,QAAQ;GAE3E,MAAM,SAAoB;IACxB;IACA,cAAc,KAAK;IACnB,aAAa,KAAK,QAAQ;IAC1B,MAAM,KAAK;IACX,OAAO,QAAQ;IACf,YAAY,QAAQ;IACpB,YAAY,QAAQ;IACpB,WAAW,KAAK;IAChB;IACA,SAAS,QAAQ;IACjB,gBAAgB,QAAQ;IACxB,YAAY,QAAQ;IACpB,iBAAiB;IACjB,eAAe,QAAQ;IACvB,SAAS;IACT,GAAI,QAAQ,eAAe,EAAE,cAAc,QAAQ,aAAa,IAAI,CAAC;IACrE,aAAa,QAAQ;IACrB;IACA,YAAY,KAAK,SAAS;GAC5B;GACA,MAAM,gBACJ,QAAQ,iBACP,OAAO,KAAK,iBAAiB,aAC1B,MAAM,KAAK,aAAa;IACtB,YAAY,KAAK;IACjB;IACA,WAAW,KAAK,QAAQ;IACxB,YAAY,KAAK,SAAS;IAC1B,MAAM,KAAK;IACX,SAAS,KAAK,QAAQ;IACtB,cAAc,KAAK,SAAS,QAAQ,CAAC;GACvC,CAAC,IACD,KAAK;GACX,IAAI,kBAAkB,KAAA,GAAW;IAC/B,MAAM,eAAe,MAAM,wBAAwB,aAAa;IAChE,6BAA6B,cAAc,QAAQ,OAAO,QAAQ,UAAU;IAC5E,OAAO,eAAe;GACxB;GACA,OAAO;IAAE,QAAQ,kBAAkB,MAAM;IAAG,WAAW;GAAgB;EACzE,UAAU;GAIR,SAAS,OAAO,KAAK;EACvB;CACF;CAEA,MAAM,UAAU,MAAM,KAAK,EAAE,QAAQ,KAAK,IAAI,aAAa,MAAM,MAAM,EAAE,SAAS,OAAO,CAAC;CAI1F,MAAM,QAAQ,WAAW,OAAO;CAMhC,KAAK,MAAM,CAAC,OAAO,YAAY,UAC7B,MAAM,cAAc,SAAS,OAAO,kDAAkD;CAExF,SAAS,MAAM;CAEf,IAAI,cAAc,SAAS,GACzB,MAAM,cAAc,WAAW,IAC3B,cAAc,KACd,IAAI,eACF,eACA,oBAAoB,cAAc,OAAO,iDAC3C;CAIN,IAAI;CACJ,IAAI,KAAK,QAQP,SAAS,MAAM,eAAe,MAAM;EANlC,GAAG,KAAK;EACR,YAAY,KAAK,OAAO;EACxB,OAAO,aAAa,QAAQ,WAAW;EACvC,aAAa,IAAI,KAAK,IAAI,CAAC,CAAC,CAAC,YAAY;EACzC,qBAAqB,uBAAuB,KAAA;CAED,CAAC;CAGhD,MAAM,UAAU,IAAI,KAAK,IAAI,CAAC,CAAC,CAAC,YAAY;CAE5C,OAAO;EACL,YAAY,KAAK;EACjB;EACA;EACA;EACA;EACA;EACA;EACA;EACA;CACF;AACF;AAIA,IAAM,qBAAN,cAAiC,MAAM;CACrC;CACA;CACA,YAAY,QAAmB,WAAgC;EAC7D,MAAM,QAAQ,OAAO,UAAU,GAAG,OAAO,WAAW,GAAG,OAAO,KAAK,WAAW,OAAO,QAAQ;EAC7F,KAAK,SAAS;EACd,KAAK,YAAY;CACnB;AACF;;;;;;;;;;;;;;;;AAiBA,eAAsB,cACpB,SACA,OACA,QACe;CACf,MAAM,WAAW,MAAM,QAAQ,WAAW,OAAO,KAAK;CACtD,IAAI,aAAa,KAAA,GAAW;CAC5B,IAAI,SAAS,WAAW,WAAW;CACnC,MAAM,QAAQ,SAAS,MAAM;AAC/B;AAEA,SAAS,sBAAsB,SAA6B;CAC1D,QAAQ,WAAmD;EACzD,IAAI,CAAC,SACH,MAAM,IAAI,MACR,6LACF;EAEF,OAAO,IAAI,0BAA0B,EACnC,KAAK,GAAG,QAAQ,cAAc,OAAO,QACvC,CAAC;CACH;AACF;AAEA,eAAe,wBACb,OAC2B;CAC3B,IAAI,mBAAmB,KAAK,GAAG;EAC7B,IAAI,CAAE,MAAM,uBAAuB,KAAK,GACtC,MAAM,IAAI,MAAM,iEAAiE;EAEnF,OAAO;CACT;CACA,OAAO,sBAAsB,KAAK;AACpC;AAEA,SAAS,mBACP,OAC2B;CAC3B,OAAO,mBAAmB,SAAS,YAAY;AACjD;AAEA,SAAS,6BACP,SACA,OACA,YACM;CACN,IAAI,QAAQ,UAAU,KAAA,KAAa,QAAQ,UAAU,OACnD,MAAM,IAAI,MACR,wCAAwC,QAAQ,MAAM,kCAAkC,MAAM,EAChG;CAEF,IAAI,QAAQ,eAAe,KAAA,KAAa,QAAQ,eAAe,YAC7D,MAAM,IAAI,MACR,6CAA6C,QAAQ,WAAW,uCAAuC,WAAW,EACpH;AAEJ;AAEA,SAAS,aAAa,QAAuC;CAG3D,MAAM,OAAO,GAAG,OAAO,WAAW,IAAI,OAAO,UAAU,IAAI,OAAO,WAAW,IAAI,OAAO;CAMxF,IAAI,KAAK;CACT,IAAI,KAAK;CACT,KAAK,IAAI,IAAI,GAAG,IAAI,KAAK,QAAQ,KAAK;EACpC,MAAM,IAAI,KAAK,WAAW,CAAC;EAC3B,KAAK,KAAK,KAAK,KAAK,GAAG,QAAU,MAAM;EACvC,KAAK,KAAK,KAAK,KAAK,GAAG,UAAU,MAAM;CACzC;CACA,OAAO,OAAO,GAAG,SAAS,EAAE,CAAC,CAAC,SAAS,GAAG,GAAG,IAAI,GAAG,SAAS,EAAE,CAAC,CAAC,SAAS,GAAG,GAAG;AAClF"}
|
|
@@ -1,8 +1,8 @@
|
|
|
1
1
|
import { c as ValidationError, r as CaptureIntegrityError } from "../errors-DEE6u6ot.js";
|
|
2
2
|
import { n as bonferroni, r as holm, t as benjaminiHochberg } from "../multiplicity-DIWHvysC.js";
|
|
3
|
-
import { $ as paretoSignificanceGate, A as PairArmsOptions, B as PowerPreflight, C as hashJson, E as verifyManifest, H as powerPreflight, K as EvidenceVector, L as comparePairedArms, N as PairedArmRow, O as MatchedPair, Q as paretoPolicy, R as pairArms, S as evaluateHypothesis, T as signManifest, V as PowerPreflightOptions, X as PromotionPolicy, Z as buildEvidenceVector, _ as sequentialPairedGate, _t as
|
|
3
|
+
import { $ as paretoSignificanceGate, A as PairArmsOptions, B as PowerPreflight, C as hashJson, Dt as EProcessStep, E as verifyManifest, Et as EProcessState, Ft as mcnemar, H as powerPreflight, K as EvidenceVector, L as comparePairedArms, Lt as pairedRiskDifference, N as PairedArmRow, O as MatchedPair, Ot as eProcess, Q as paretoPolicy, R as pairArms, Rt as pairedRiskDifferenceExact, S as evaluateHypothesis, T as signManifest, Tt as EProcessOptions, V as PowerPreflightOptions, Vt as wilson, X as PromotionPolicy, Z as buildEvidenceVector, _ as sequentialPairedGate, _t as verifyEvidenceReceipt, at as createCampaignEvidenceReceipt, b as SignedManifest, c as pairHoldout, ct as EVIDENCE_RECEIPT_VERSION, d as SequentialDecision, dt as EvidenceBinding, f as SequentialObservation, ft as EvidenceReceipt, g as sequentialDecide, gt as isIndependentEvidence, ht as createEvidenceReceipt, i as PairedHoldout, it as CampaignEvidenceContext, j as PairArmsResult, jt as ProportionInterval, lt as EvidenceAuthority, m as SequentialPairedGateOptions, mt as INDEPENDENT_EVIDENCE_AUTHORITY_KINDS, n as HeldoutSignificance, ot as CreateEvidenceReceiptInput, p as SequentialPairedGate, pt as EvidenceReceiptVerification, r as HeldoutSignificanceOptions, s as heldoutSignificance, st as EVIDENCE_AUTHORITY_KINDS, ut as EvidenceAuthorityKind, v as HypothesisManifest, w as manifestContentDigest, wt as EProcess, x as SignedManifestAlgo, y as HypothesisResult, z as pairRunRecords, zt as pairedRiskDifferenceScore } from "../statistical-heldout-Z9NROFFS.js";
|
|
4
4
|
import { c as BOOTSTRAP_GATE_MIN_N, d as PairedBootstrapResult, h as pairedBootstrap, u as PairedBootstrapOptions } from "../paired-promotion-decision-CGzg0cI_.js";
|
|
5
|
-
import {
|
|
5
|
+
import { _ as requiredSampleSize, a as ExperimentVerdict, d as inMemoryExperimentStore, f as mulberry32, g as requiredPairedSampleSize, h as pairedMde, i as ExperimentTracker, l as fileExperimentStore, m as mcnemarRequiredN, n as ExperimentRep, p as mcnemarPower, r as ExperimentStats, t as Experiment } from "../experiment-tracker-CNwqCZFD.js";
|
|
6
6
|
import { a as PairedEvalueStep, d as sequentialCrossingHorizon, i as PairedEvalueSequence, o as SequentialCrossingHorizon, r as PairedEvalueOptions, s as SequentialCrossingHorizonOptions, u as pairedEvalueSequence } from "../sequential-BhsrMupG.js";
|
|
7
7
|
import { z } from "zod";
|
|
8
8
|
//#region src/experiment/ast.d.ts
|
|
@@ -758,71 +758,6 @@ interface RegisteredExperiment {
|
|
|
758
758
|
*/
|
|
759
759
|
declare function openSealedExperiment(sealed: SealedExperiment): Promise<RegisteredExperiment>;
|
|
760
760
|
//#endregion
|
|
761
|
-
//#region src/experiment/evidence-receipt.d.ts
|
|
762
|
-
declare const EVIDENCE_RECEIPT_VERSION: '1.0.0';
|
|
763
|
-
/** Closed promotion vocabulary. A typo or unknown future kind is never independent by default. */
|
|
764
|
-
declare const EVIDENCE_AUTHORITY_KINDS: readonly ['candidate-self-report', 'independent-evaluator', 'independent-replication', 'human-review', 'production-canary'];
|
|
765
|
-
type EvidenceAuthorityKind = (typeof EVIDENCE_AUTHORITY_KINDS)[number];
|
|
766
|
-
declare const INDEPENDENT_EVIDENCE_AUTHORITY_KINDS: readonly ["independent-evaluator", "independent-replication", "human-review", "production-canary"];
|
|
767
|
-
interface EvidenceAuthority {
|
|
768
|
-
readonly kind: EvidenceAuthorityKind;
|
|
769
|
-
/** Stable authority/service/person identifier; never inferred from model prose. */
|
|
770
|
-
readonly id: string;
|
|
771
|
-
}
|
|
772
|
-
interface EvidenceBinding {
|
|
773
|
-
readonly schemaVersion: typeof EVIDENCE_RECEIPT_VERSION;
|
|
774
|
-
/** Stable objective identity spanning retries, forks, resumes, and multiple runs. */
|
|
775
|
-
readonly pursuitId: string;
|
|
776
|
-
/** Concrete execution/evaluation run this evidence was produced from. */
|
|
777
|
-
readonly runId: string;
|
|
778
|
-
/** Exact candidate/program/profile bundle under evaluation. */
|
|
779
|
-
readonly candidateDigest: string;
|
|
780
|
-
/** Exact evaluator/checker program or registered experiment definition. */
|
|
781
|
-
readonly evaluatorDigest: string;
|
|
782
|
-
/** Exact sandbox/container/dependency/environment identity. */
|
|
783
|
-
readonly environmentDigest: string;
|
|
784
|
-
/** Commitment to the input set. The hidden inputs themselves need not be exposed. */
|
|
785
|
-
readonly inputSetCommitment: string;
|
|
786
|
-
/** Content identity of the raw evaluated output/artifact set. */
|
|
787
|
-
readonly outputDigest: string;
|
|
788
|
-
/** Content identity of the evaluation result/report/verdict body. */
|
|
789
|
-
readonly resultDigest: string;
|
|
790
|
-
readonly authority: EvidenceAuthority;
|
|
791
|
-
/** Optional experiment seal that governed the evaluation. */
|
|
792
|
-
readonly experimentDigest?: string;
|
|
793
|
-
/** Optional observer-journal chain tip covering the execution being evaluated. */
|
|
794
|
-
readonly observerDigest?: string;
|
|
795
|
-
}
|
|
796
|
-
interface EvidenceReceipt {
|
|
797
|
-
readonly binding: EvidenceBinding;
|
|
798
|
-
readonly attestation: AttestedReport;
|
|
799
|
-
}
|
|
800
|
-
interface EvidenceReceiptVerification {
|
|
801
|
-
readonly valid: boolean;
|
|
802
|
-
readonly reason?: string;
|
|
803
|
-
}
|
|
804
|
-
interface CreateEvidenceReceiptInput extends Omit<EvidenceBinding, 'schemaVersion'> {}
|
|
805
|
-
/**
|
|
806
|
-
* Mint a content-attested evidence receipt. Required identity fields are deliberately
|
|
807
|
-
* non-optional: unknown evidence stays unknown and cannot accidentally look certified.
|
|
808
|
-
* The attestation's input provenance must equal the receipt commitment so two competing
|
|
809
|
-
* descriptions of the evaluated population cannot coexist inside one valid receipt.
|
|
810
|
-
*/
|
|
811
|
-
declare function createEvidenceReceipt(input: CreateEvidenceReceiptInput, provenance: AttestationProvenance): EvidenceReceipt;
|
|
812
|
-
/**
|
|
813
|
-
* Verify promotion-grade evidence. Generic report attestation keeps a legacy read path,
|
|
814
|
-
* but an EvidenceReceipt never accepts unbound provenance: changing the evaluator code,
|
|
815
|
-
* model versions, input commitment provenance, or creation record must invalidate the
|
|
816
|
-
* evidence rather than merely annotating it as legacy.
|
|
817
|
-
*/
|
|
818
|
-
declare function verifyEvidenceReceipt(receipt: EvidenceReceipt): EvidenceReceiptVerification;
|
|
819
|
-
/**
|
|
820
|
-
* Promotion may choose a stricter policy, but this primitive makes the basic separation
|
|
821
|
-
* explicit: only a recognized independent authority is independent. Unknown/forged kinds
|
|
822
|
-
* and candidate self-reports both return false.
|
|
823
|
-
*/
|
|
824
|
-
declare function isIndependentEvidence(receipt: EvidenceReceipt): boolean;
|
|
825
|
-
//#endregion
|
|
826
761
|
//#region src/experiment/evidence-record.d.ts
|
|
827
762
|
/**
|
|
828
763
|
* Trust ladder for a recorded claim, strongest first.
|
|
@@ -970,5 +905,5 @@ declare function clusteredPower(options: ClusteredPowerOptions): ClusteredPowerR
|
|
|
970
905
|
/** Throw the refusal for callers that want configuration-time failure. */
|
|
971
906
|
declare function assertDesignAdequate(result: ClusteredPowerResult): void;
|
|
972
907
|
//#endregion
|
|
973
|
-
export { type AdmissionExecution, type AdmissionPartition, type AdmissionRule, type AdmissionStage, type ArmRealizedBudget, type ArmSpec, BOOTSTRAP_GATE_MIN_N, type BudgetRule, type ClusteredPowerOptions, type ClusteredPowerPoint, type ClusteredPowerRefusal, type ClusteredPowerResult, type ComputedInterval, type Condition, type CreateEvidenceReceiptInput, type DecisionBranch, type DecisionOutcome, type DecisionRule, type DerivedQuantities, DesignRefusalError, type EProcess, type EProcessOptions, type EProcessState, type EProcessStep, EVIDENCE_AUTHORITY_KINDS, EVIDENCE_RECEIPT_VERSION, EVIDENCE_STATES, type Estimand, type EstimandResult, type EvidenceAuthority, type EvidenceAuthorityKind, type EvidenceBinding, type EvidenceDenominator, type EvidenceReceipt, type EvidenceReceiptVerification, type EvidenceRecord, type EvidenceRegistryRecord, type EvidenceState, type EvidenceVector, type Experiment, type ExperimentFunnel, type ExperimentRep, type ExperimentSpec, type ExperimentStats, ExperimentTracker, type ExperimentVerdict, FunnelIntegrityError, type FunnelPartitionCount, type FunnelStageCount, type FunnelStageInput, type GateEvidence, type GateResult, type HaltOutcome, type HaltRule, type HeldoutSignificance, type HeldoutSignificanceOptions, type HypothesisManifest, type HypothesisResult, INDEPENDENT_EVIDENCE_AUTHORITY_KINDS, type IntervalSpec, type JsonValue, MatchedBudgetError, type MatchedBudgetRule, type MatchedBudgetVerdict, type MatchedPair, type NLadderProjection, type Obligation, type OutcomeSpec, type PairArmsOptions, type PairArmsResult, type PairedArmRow, type PairedBootstrapOptions, type PairedBootstrapResult, type PairedEvalueOptions, type PairedEvalueSequence, type PairedEvalueStep, type PairedHoldout, type PowerPreflight, type PowerPreflightOptions, type Predicate, type PromotionPolicy, type ProportionInterval, type RegisteredExperiment, type ReissuePolicy, type ReissueVerdict, type SealAmendment, SealIntegrityError, type SealedExperiment, type SeedDerivation, type SelectionRule, type SequentialCrossingHorizon, type SequentialCrossingHorizonOptions, type SequentialDecision, type SequentialObservation, type SequentialPairedGate, type SequentialPairedGateOptions, type SetExpr, type SignFlipFloor, type SignedManifest, type SignedManifestAlgo, type UniformPassDecision, type UniformPassSchedule, type ValidityGate, amendExperiment, assertDesignAdequate, assertFunnelReconciles, assertMatchedBudgets, benjaminiHochberg, bonferroni, buildEvidenceVector, buildFunnel, classifyReissue, clusteredPower, comparePairedArms, composeFunnels, computeEstimand, computeInterval, createEvidenceReceipt, defineExperiment, eProcess, evaluateCondition, evaluateHaltRule, evaluateHypothesis, evaluateIdentityGate, evaluateOracleDeterminismGate, evaluatePopulationReproducibilityGate, evaluatePowerFloorGate, evaluatePredicate, evaluateProvenanceGate, evidenceRegistryRecordSchema, executeAdmissionRule, executeDecisionRule, fileExperimentStore, hashJson, heldoutSignificance, holm, inMemoryExperimentStore, isIndependentEvidence, manifestContentDigest, mcnemar, mcnemarPower, mcnemarRequiredN, mulberry32, openSealedExperiment, pairArms, pairHoldout, pairRunRecords, pairedBootstrap, pairedEvalueSequence, pairedMde, pairedRiskDifference, pairedRiskDifferenceExact, pairedRiskDifferenceScore, paretoPolicy, paretoSignificanceGate, parseEvidenceRegistryRecord, powerPreflight, projectNLadderBudget, readField, renderEvidenceIndex, renderFunnelTable, requiredPairedSampleSize, requiredSampleSize, runSelectionRule, runUniformPassBudget, sealExperiment, sequentialCrossingHorizon, sequentialDecide, sequentialPairedGate, signManifest, validateEvidenceRegistry, verifyEvidenceReceipt, verifyManifest, verifyMatchedBudgets, verifySealedExperiment, wilson };
|
|
908
|
+
export { type AdmissionExecution, type AdmissionPartition, type AdmissionRule, type AdmissionStage, type ArmRealizedBudget, type ArmSpec, BOOTSTRAP_GATE_MIN_N, type BudgetRule, type CampaignEvidenceContext, type ClusteredPowerOptions, type ClusteredPowerPoint, type ClusteredPowerRefusal, type ClusteredPowerResult, type ComputedInterval, type Condition, type CreateEvidenceReceiptInput, type DecisionBranch, type DecisionOutcome, type DecisionRule, type DerivedQuantities, DesignRefusalError, type EProcess, type EProcessOptions, type EProcessState, type EProcessStep, EVIDENCE_AUTHORITY_KINDS, EVIDENCE_RECEIPT_VERSION, EVIDENCE_STATES, type Estimand, type EstimandResult, type EvidenceAuthority, type EvidenceAuthorityKind, type EvidenceBinding, type EvidenceDenominator, type EvidenceReceipt, type EvidenceReceiptVerification, type EvidenceRecord, type EvidenceRegistryRecord, type EvidenceState, type EvidenceVector, type Experiment, type ExperimentFunnel, type ExperimentRep, type ExperimentSpec, type ExperimentStats, ExperimentTracker, type ExperimentVerdict, FunnelIntegrityError, type FunnelPartitionCount, type FunnelStageCount, type FunnelStageInput, type GateEvidence, type GateResult, type HaltOutcome, type HaltRule, type HeldoutSignificance, type HeldoutSignificanceOptions, type HypothesisManifest, type HypothesisResult, INDEPENDENT_EVIDENCE_AUTHORITY_KINDS, type IntervalSpec, type JsonValue, MatchedBudgetError, type MatchedBudgetRule, type MatchedBudgetVerdict, type MatchedPair, type NLadderProjection, type Obligation, type OutcomeSpec, type PairArmsOptions, type PairArmsResult, type PairedArmRow, type PairedBootstrapOptions, type PairedBootstrapResult, type PairedEvalueOptions, type PairedEvalueSequence, type PairedEvalueStep, type PairedHoldout, type PowerPreflight, type PowerPreflightOptions, type Predicate, type PromotionPolicy, type ProportionInterval, type RegisteredExperiment, type ReissuePolicy, type ReissueVerdict, type SealAmendment, SealIntegrityError, type SealedExperiment, type SeedDerivation, type SelectionRule, type SequentialCrossingHorizon, type SequentialCrossingHorizonOptions, type SequentialDecision, type SequentialObservation, type SequentialPairedGate, type SequentialPairedGateOptions, type SetExpr, type SignFlipFloor, type SignedManifest, type SignedManifestAlgo, type UniformPassDecision, type UniformPassSchedule, type ValidityGate, amendExperiment, assertDesignAdequate, assertFunnelReconciles, assertMatchedBudgets, benjaminiHochberg, bonferroni, buildEvidenceVector, buildFunnel, classifyReissue, clusteredPower, comparePairedArms, composeFunnels, computeEstimand, computeInterval, createCampaignEvidenceReceipt, createEvidenceReceipt, defineExperiment, eProcess, evaluateCondition, evaluateHaltRule, evaluateHypothesis, evaluateIdentityGate, evaluateOracleDeterminismGate, evaluatePopulationReproducibilityGate, evaluatePowerFloorGate, evaluatePredicate, evaluateProvenanceGate, evidenceRegistryRecordSchema, executeAdmissionRule, executeDecisionRule, fileExperimentStore, hashJson, heldoutSignificance, holm, inMemoryExperimentStore, isIndependentEvidence, manifestContentDigest, mcnemar, mcnemarPower, mcnemarRequiredN, mulberry32, openSealedExperiment, pairArms, pairHoldout, pairRunRecords, pairedBootstrap, pairedEvalueSequence, pairedMde, pairedRiskDifference, pairedRiskDifferenceExact, pairedRiskDifferenceScore, paretoPolicy, paretoSignificanceGate, parseEvidenceRegistryRecord, powerPreflight, projectNLadderBudget, readField, renderEvidenceIndex, renderFunnelTable, requiredPairedSampleSize, requiredSampleSize, runSelectionRule, runUniformPassBudget, sealExperiment, sequentialCrossingHorizon, sequentialDecide, sequentialPairedGate, signManifest, validateEvidenceRegistry, verifyEvidenceReceipt, verifyManifest, verifyMatchedBudgets, verifySealedExperiment, wilson };
|
|
974
909
|
//# sourceMappingURL=index.d.ts.map
|