@tangle-network/agent-eval 0.173.3 → 0.174.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (125) hide show
  1. package/CHANGELOG.md +31 -0
  2. package/dist/{proposal-findings-bko3GGy-.js → abort-signal-CtzAM_sJ.js} +11 -11
  3. package/dist/abort-signal-CtzAM_sJ.js.map +1 -0
  4. package/dist/adapters/http.d.ts +2 -2
  5. package/dist/agent-profile-_xPxqVJt.d.ts +488 -0
  6. package/dist/agent-profile-_xPxqVJt.d.ts.map +1 -0
  7. package/dist/analyst/index.d.ts +7 -9
  8. package/dist/analyst/index.d.ts.map +1 -1
  9. package/dist/analyst/index.js +8 -8
  10. package/dist/{benchmark-C4wk_Sjr.js → benchmark-DQKzykkO.js} +2 -2
  11. package/dist/{benchmark-C4wk_Sjr.js.map → benchmark-DQKzykkO.js.map} +1 -1
  12. package/dist/{benchmark-command-BY9oscke.js → benchmark-command-mZIlR-ra.js} +13 -13
  13. package/dist/{benchmark-command-BY9oscke.js.map → benchmark-command-mZIlR-ra.js.map} +1 -1
  14. package/dist/benchmarks/index.d.ts +3 -4
  15. package/dist/benchmarks/index.d.ts.map +1 -1
  16. package/dist/benchmarks/index.js +3 -3
  17. package/dist/campaign/index.d.ts +5 -9
  18. package/dist/campaign/index.js +7 -7
  19. package/dist/{campaign-B3kPMU8S.js → campaign-BzMSCejE.js} +8 -8
  20. package/dist/{campaign-B3kPMU8S.js.map → campaign-BzMSCejE.js.map} +1 -1
  21. package/dist/cli.js +1 -1
  22. package/dist/{client-DlqdbM7n.d.ts → client-vyYQg3bm.d.ts} +2 -2
  23. package/dist/{client-DlqdbM7n.d.ts.map → client-vyYQg3bm.d.ts.map} +1 -1
  24. package/dist/contract/index.d.ts +9 -10
  25. package/dist/contract/index.js +8 -8
  26. package/dist/{default-registry-B0bKikCb.js → default-registry-CrAp0pYq.js} +4 -4
  27. package/dist/{default-registry-B0bKikCb.js.map → default-registry-CrAp0pYq.js.map} +1 -1
  28. package/dist/{default-registry-BKwc8bN5.d.ts → default-registry-FfNzaUHV.d.ts} +3 -3
  29. package/dist/{default-registry-BKwc8bN5.d.ts.map → default-registry-FfNzaUHV.d.ts.map} +1 -1
  30. package/dist/{define-agent-eval-CY6qdlGV.d.ts → define-agent-eval-V1jQyCDR.d.ts} +102 -11
  31. package/dist/define-agent-eval-V1jQyCDR.d.ts.map +1 -0
  32. package/dist/{define-agent-eval-8h3lXXee.js → define-agent-eval-ox5McL6e.js} +331 -144
  33. package/dist/define-agent-eval-ox5McL6e.js.map +1 -0
  34. package/dist/{dspy-rlm-engine-CF0t2ITD.js → dspy-rlm-engine-Caz2pl4L.js} +3 -3
  35. package/dist/{dspy-rlm-engine-CF0t2ITD.js.map → dspy-rlm-engine-Caz2pl4L.js.map} +1 -1
  36. package/dist/{engine-DhFir3Ys.d.ts → engine-CvW_I72-.d.ts} +2 -2
  37. package/dist/{engine-DhFir3Ys.d.ts.map → engine-CvW_I72-.d.ts.map} +1 -1
  38. package/dist/experiment/index.d.ts +1 -4
  39. package/dist/experiment/index.d.ts.map +1 -1
  40. package/dist/{external-optimizer-process-BwITA9Jp.js → external-optimizer-process-CxnFL1hd.js} +2 -2
  41. package/dist/{external-optimizer-process-BwITA9Jp.js.map → external-optimizer-process-CxnFL1hd.js.map} +1 -1
  42. package/dist/{external-optimizer-subprocess-wBWeoG6A.js → external-optimizer-subprocess-CQi27uEI.js} +2 -2
  43. package/dist/{external-optimizer-subprocess-wBWeoG6A.js.map → external-optimizer-subprocess-CQi27uEI.js.map} +1 -1
  44. package/dist/fuzz.js +1 -1
  45. package/dist/fuzz.js.map +1 -1
  46. package/dist/hosted/index.d.ts +1 -1
  47. package/dist/{index-BQqOjerE.d.ts → index-BTrx5s8m.d.ts} +8 -9
  48. package/dist/index-BTrx5s8m.d.ts.map +1 -0
  49. package/dist/{index-D0Db5X-4.d.ts → index-Bn-nlnSV.d.ts} +4 -4
  50. package/dist/{index-D0Db5X-4.d.ts.map → index-Bn-nlnSV.d.ts.map} +1 -1
  51. package/dist/index-DKXuBPXf.d.ts +3840 -0
  52. package/dist/index-DKXuBPXf.d.ts.map +1 -0
  53. package/dist/index.d.ts +11 -13
  54. package/dist/index.d.ts.map +1 -1
  55. package/dist/index.js +10 -10
  56. package/dist/{kind-factory-gP6lDySe.js → kind-factory-BLvL-E44.js} +2 -2
  57. package/dist/{kind-factory-gP6lDySe.js.map → kind-factory-BLvL-E44.js.map} +1 -1
  58. package/dist/{llm-judge-BfqMFo4h.js → llm-judge-DmNaBrXB.js} +2541 -2435
  59. package/dist/llm-judge-DmNaBrXB.js.map +1 -0
  60. package/dist/{matrix-DGu8KhSs.d.ts → matrix-CJtXz1ky.d.ts} +2 -2
  61. package/dist/{matrix-DGu8KhSs.d.ts.map → matrix-CJtXz1ky.d.ts.map} +1 -1
  62. package/dist/multishot/golden/index.d.ts +1 -1
  63. package/dist/multishot/index.d.ts +2 -2
  64. package/dist/openapi.json +1 -1
  65. package/dist/{produced-state-D91uDvQw.js → produced-state-B8mw6zj9.js} +2 -2
  66. package/dist/{produced-state-D91uDvQw.js.map → produced-state-B8mw6zj9.js.map} +1 -1
  67. package/dist/rl.d.ts +1 -1
  68. package/dist/rl.d.ts.map +1 -1
  69. package/dist/rl.js.map +1 -1
  70. package/dist/{semantic-concept-judge-Dok7_35a.js → semantic-concept-judge-E3s_fEjB.js} +3 -3
  71. package/dist/{semantic-concept-judge-Dok7_35a.js.map → semantic-concept-judge-E3s_fEjB.js.map} +1 -1
  72. package/dist/{skillopt-optimization-method-DDw3v3gA.js → skillopt-optimization-method-f7399oGb.js} +5 -5
  73. package/dist/{skillopt-optimization-method-DDw3v3gA.js.map → skillopt-optimization-method-f7399oGb.js.map} +1 -1
  74. package/dist/statistical-heldout-Cqb73yE9.d.ts +1127 -0
  75. package/dist/statistical-heldout-Cqb73yE9.d.ts.map +1 -0
  76. package/dist/{store-otlp-Dow0pk_5.js → store-otlp-DV_H2HDu.js} +2 -2
  77. package/dist/{store-otlp-Dow0pk_5.js.map → store-otlp-DV_H2HDu.js.map} +1 -1
  78. package/dist/{store-tool-spans-CCZNsihA.d.ts → store-tool-spans-4o55ABER.d.ts} +3 -3
  79. package/dist/{store-tool-spans-CCZNsihA.d.ts.map → store-tool-spans-4o55ABER.d.ts.map} +1 -1
  80. package/dist/{store-tool-spans-CeNj_m2L.js → store-tool-spans-B9tjys_h.js} +3 -3
  81. package/dist/{store-tool-spans-CeNj_m2L.js.map → store-tool-spans-B9tjys_h.js.map} +1 -1
  82. package/dist/{task-failure-attributes-CZjZeBsY.js → task-failure-attributes-CUy9mkIY.js} +2 -2
  83. package/dist/{task-failure-attributes-CZjZeBsY.js.map → task-failure-attributes-CUy9mkIY.js.map} +1 -1
  84. package/dist/{tool-groups-Cp4Xdzrp.d.ts → tool-groups-DAe1t6zb.d.ts} +2 -2
  85. package/dist/tool-groups-DAe1t6zb.d.ts.map +1 -0
  86. package/dist/trace-repair/index.d.ts +1 -1
  87. package/dist/traces.d.ts +2 -2
  88. package/dist/traces.js +4 -4
  89. package/dist/{types-Ba5UQyVD.d.ts → types-BJz2CPTM.d.ts} +2 -2
  90. package/dist/{types-Ba5UQyVD.d.ts.map → types-BJz2CPTM.d.ts.map} +1 -1
  91. package/dist/{types-CiWITkGo.js → types-DQ0e2E7y.js} +2 -2
  92. package/dist/types-DQ0e2E7y.js.map +1 -0
  93. package/dist/{types-BDV4PiMR.d.ts → types-Dd1ejaeI.d.ts} +2 -2
  94. package/dist/{types-BDV4PiMR.d.ts.map → types-Dd1ejaeI.d.ts.map} +1 -1
  95. package/docs/campaign-proposers.md +42 -0
  96. package/package.json +1 -1
  97. package/dist/agent-profile-B9_GGsG8.d.ts +0 -84
  98. package/dist/agent-profile-B9_GGsG8.d.ts.map +0 -1
  99. package/dist/backend-integrity-CeuTgqsd.d.ts +0 -280
  100. package/dist/backend-integrity-CeuTgqsd.d.ts.map +0 -1
  101. package/dist/benchmark-BjLGkfnN.d.ts +0 -236
  102. package/dist/benchmark-BjLGkfnN.d.ts.map +0 -1
  103. package/dist/define-agent-eval-8h3lXXee.js.map +0 -1
  104. package/dist/define-agent-eval-CY6qdlGV.d.ts.map +0 -1
  105. package/dist/external-optimizer-contracts-CQCpyrIL.d.ts +0 -172
  106. package/dist/external-optimizer-contracts-CQCpyrIL.d.ts.map +0 -1
  107. package/dist/heldout-gate-Df5hsqmm.d.ts +0 -453
  108. package/dist/heldout-gate-Df5hsqmm.d.ts.map +0 -1
  109. package/dist/index-BQqOjerE.d.ts.map +0 -1
  110. package/dist/index-CFDffsKz.d.ts +0 -1135
  111. package/dist/index-CFDffsKz.d.ts.map +0 -1
  112. package/dist/llm-judge-BfqMFo4h.js.map +0 -1
  113. package/dist/power-preflight-Ptse_Kq7.d.ts +0 -117
  114. package/dist/power-preflight-Ptse_Kq7.d.ts.map +0 -1
  115. package/dist/pre-registration-BoI4ucR3.d.ts +0 -592
  116. package/dist/pre-registration-BoI4ucR3.d.ts.map +0 -1
  117. package/dist/promotion-policy-CvMda3kU.d.ts +0 -134
  118. package/dist/promotion-policy-CvMda3kU.d.ts.map +0 -1
  119. package/dist/proposal-findings-bko3GGy-.js.map +0 -1
  120. package/dist/provenance-CRY67X50.d.ts +0 -1995
  121. package/dist/provenance-CRY67X50.d.ts.map +0 -1
  122. package/dist/statistical-heldout-DTyB_6-1.d.ts +0 -295
  123. package/dist/statistical-heldout-DTyB_6-1.d.ts.map +0 -1
  124. package/dist/tool-groups-Cp4Xdzrp.d.ts.map +0 -1
  125. package/dist/types-CiWITkGo.js.map +0 -1
@@ -1,1135 +0,0 @@
1
- import { c as ValidationError, t as AgentEvalError } from "./errors-DEE6u6ot.js";
2
- import { T as RunPaidCallInput, p as CostProvenance } from "./cost-ledger-DbQdN3nO.js";
3
- import { a as RunRecord, s as RunSplitTag } from "./run-record-DTv1MdjK.js";
4
- import { f as AnalystUsageReceipt, i as AnalystFinding, l as AnalystRunResult, w as TraceAnalysisStore } from "./types-DN2WdT5S.js";
5
- import { _ as PairedArmsComparison } from "./pre-registration-BoI4ucR3.js";
6
- import { d as PairedBootstrapResult } from "./paired-promotion-decision-CGzg0cI_.js";
7
- import { A as LabeledScenarioWrite, D as LabeledScenarioSampleArgs, E as LabeledScenarioRecord, O as LabeledScenarioSource, R as Scenario, S as JudgeConfig, T as LabelTrust, a as CampaignResult, d as DispatchContext, j as MutableSurface, k as LabeledScenarioStore, l as CodeSurface, p as Gate, u as ComponentSurface } from "./types-Ba5UQyVD.js";
8
- import "./heldout-gate-Df5hsqmm.js";
9
- import { In as CampaignRunPlan, Kn as CampaignStorage, Rn as PlanCampaignRunOptions } from "./provenance-CRY67X50.js";
10
- import "./promotion-policy-CvMda3kU.js";
11
- import { d as CompletionVerdict, f as CorrectnessChecker, h as ProducedState, n as BackendIntegrityReport, s as RuntimeEventLike, u as CompletionRequirement } from "./backend-integrity-CeuTgqsd.js";
12
- import { _ as AnalystIssueExpectation, a as AnalystBenchmarkLabelState, t as AnalystBenchmarkCase } from "./benchmark-BjLGkfnN.js";
13
- import { t as AgentProfile$1 } from "./agent-profile-B9_GGsG8.js";
14
- import "./statistical-heldout-DTyB_6-1.js";
15
- import { AgentProfile } from "@tangle-network/agent-interface";
16
- //#region src/campaign/search-ledger-errors.d.ts
17
- /** Base error for invalid search-ledger input or operations. */
18
- declare class SearchLedgerError extends ValidationError {}
19
- /** Error raised when durable search-ledger data fails an integrity check. */
20
- declare class SearchLedgerIntegrityError extends SearchLedgerError {}
21
- /** Error raised when an event identifier is reused with different content. */
22
- declare class SearchLedgerConflictError extends SearchLedgerError {}
23
- //#endregion
24
- //#region src/campaign/analyst-surface.d.ts
25
- interface TraceAnalystScenario extends Scenario {
26
- kind: 'trace-analyst';
27
- traceStore: TraceAnalysisStore;
28
- labelState: AnalystBenchmarkLabelState;
29
- expectedIssues: readonly AnalystIssueExpectation[];
30
- labeledEvidence?: AnalystBenchmarkCase['labeledEvidence'];
31
- }
32
- interface TraceAnalystArtifact {
33
- findings: readonly AnalystFinding[];
34
- usage?: AnalystUsageReceipt;
35
- run?: AnalystRunResult;
36
- }
37
- interface BuildTraceAnalystSurfaceDispatchOptions {
38
- analyze(input: {
39
- instructions: string;
40
- traceStore: TraceAnalysisStore;
41
- runId: string;
42
- signal: AbortSignal;
43
- }): Promise<TraceAnalystArtifact>;
44
- }
45
- declare function buildTraceAnalystSurfaceDispatch(options: BuildTraceAnalystSurfaceDispatchOptions): (surface: MutableSurface, scenario: TraceAnalystScenario, context: DispatchContext) => Promise<TraceAnalystArtifact>;
46
- declare function traceAnalystQualityJudge(): JudgeConfig<TraceAnalystArtifact, TraceAnalystScenario>;
47
- //#endregion
48
- //#region src/campaign/cross-surface-types.d.ts
49
- /** Whether one candidate attempt produced a usable executable outcome. */
50
- type CrossSurfaceAttemptCompleteness = 'complete' | 'missing' | 'invalid';
51
- /** One independently proposed change on one caller-defined surface. */
52
- interface CrossSurfaceComponent {
53
- componentId: string;
54
- surfaceId: string;
55
- /** Explicitly controls whether this component may anchor the best-single arm. */
56
- bestSingleEligible: boolean;
57
- }
58
- /** Immutable identity for a single candidate or a materialized composition. */
59
- interface CrossSurfaceCandidate {
60
- candidateId: string;
61
- componentIds: string[];
62
- contentHash: string;
63
- artifactBytes: number;
64
- }
65
- /** Per-component trace evidence captured during one task attempt. */
66
- interface CrossSurfaceComponentEvidence {
67
- componentId: string;
68
- /** null means the trace could not establish whether the component fired. */
69
- fired: boolean | null;
70
- /** null means the trace could not establish whether the component changed behavior. */
71
- effectObserved: boolean | null;
72
- }
73
- /**
74
- * Canonical per-task input row. Consumers may extend this interface with
75
- * receipt, trace, retry, or failure details; the report preserves the original
76
- * row object rather than projecting those details away.
77
- */
78
- interface CrossSurfaceTaskRow {
79
- taskId: string;
80
- candidateId: string;
81
- /** Repeated here so every persisted row remains self-describing. */
82
- componentIds: string[];
83
- completeness: CrossSurfaceAttemptCompleteness;
84
- pass: boolean | null;
85
- score: number | null;
86
- /**
87
- * Per-attempt deployment measurements. Every declared metric must have a
88
- * known, non-negative value. Proposal, analysis, and selection spend belongs
89
- * in the search ledger rather than being spread across task cells.
90
- */
91
- cost: Record<string, number | null>;
92
- componentEvidence: CrossSurfaceComponentEvidence[];
93
- /** Required for missing or invalid attempts; forbidden for complete attempts. */
94
- rejectReason: string | null;
95
- }
96
- interface CrossSurfaceBootstrapPolicy {
97
- seed: number;
98
- resamples: number;
99
- confidence: number;
100
- }
101
- /** Predeclared candidate eligibility and composition policy. */
102
- interface CrossSurfaceSelectionPolicy {
103
- minimumFiringTasks: number;
104
- minimumEffectTasks: number;
105
- requireObservedFiring: boolean;
106
- requireObservedEffect: boolean;
107
- /** Only named metrics are constrained; all declared metrics are still reported. */
108
- maximumMedianCostRatioToBaseline: Record<string, number>;
109
- /** A smaller terminal bundle is reported but cannot become the selected arm. */
110
- minimumBundleComponents: number;
111
- }
112
- interface AnalyzeCrossSurfaceInteractionsInput<TRow extends CrossSurfaceTaskRow = CrossSurfaceTaskRow> {
113
- components: readonly CrossSurfaceComponent[];
114
- candidates: readonly CrossSurfaceCandidate[];
115
- rows: readonly TRow[];
116
- baselineCandidateId: string;
117
- /** Exact shared task axis and its canonical output order. */
118
- taskOrder: readonly string[];
119
- /** Canonical materialization order for component sets and the naive stack. */
120
- componentOrder: readonly string[];
121
- /** Final deterministic tie-break; lower index wins. */
122
- candidateOrder: readonly string[];
123
- /** Declares every cost key and the order used for cost tie-breaks. */
124
- costMetricOrder: readonly string[];
125
- bootstrap: CrossSurfaceBootstrapPolicy;
126
- selection: CrossSurfaceSelectionPolicy;
127
- }
128
- interface CrossSurfaceDistribution {
129
- n: number;
130
- min: number;
131
- median: number;
132
- mean: number;
133
- max: number;
134
- total: number;
135
- }
136
- interface CrossSurfaceEvidenceBreakdown {
137
- componentId: string;
138
- observedTaskIds: string[];
139
- notObservedTaskIds: string[];
140
- unobservedTaskIds: string[];
141
- }
142
- interface CrossSurfaceCandidateEvidence {
143
- byComponent: CrossSurfaceEvidenceBreakdown[];
144
- allObservedTaskIds: string[];
145
- someObservedTaskIds: string[];
146
- noneObservedTaskIds: string[];
147
- unobservedTaskIds: string[];
148
- }
149
- type CrossSurfaceIneligibilityReason = 'missing_attempt' | 'invalid_attempt' | 'baseline_outcome_missing' | 'benefit_not_greater_than_regression' | 'firing_below_minimum' | 'firing_unobserved' | 'effect_below_minimum' | 'effect_unobserved' | 'cost_limit_exceeded';
150
- interface CrossSurfaceEligibility {
151
- eligible: boolean;
152
- reasons: CrossSurfaceIneligibilityReason[];
153
- }
154
- interface CrossSurfaceCandidateOutcome {
155
- resolvedTaskIds: string[];
156
- failedTaskIds: string[];
157
- missingTaskIds: string[];
158
- invalidTaskIds: string[];
159
- benefitTaskIds: string[];
160
- regressionTaskIds: string[];
161
- comparisonMissingTaskIds: string[];
162
- netBenefit: number;
163
- }
164
- interface CrossSurfaceCandidateSummary {
165
- candidate: CrossSurfaceCandidate;
166
- outcome: CrossSurfaceCandidateOutcome;
167
- score: CrossSurfaceDistribution | null;
168
- costs: Record<string, CrossSurfaceDistribution>;
169
- firing: CrossSurfaceCandidateEvidence;
170
- effect: CrossSurfaceCandidateEvidence;
171
- /** Reuses the package's paired McNemar/risk-difference/bootstrap statistics. */
172
- comparisonToBaseline: PairedArmsComparison | null;
173
- /** null only for the fixed baseline. */
174
- eligibility: CrossSurfaceEligibility | null;
175
- }
176
- interface CrossSurfaceRelativeCost {
177
- treatmentMedian: number;
178
- comparatorMedian: number;
179
- medianDelta: number;
180
- /** null when the comparator median is zero but the treatment median is not. */
181
- medianRatio: number | null;
182
- }
183
- interface CrossSurfaceCandidateComparison {
184
- comparatorCandidateId: string;
185
- treatmentCandidateId: string;
186
- winsTaskIds: string[];
187
- regressionTaskIds: string[];
188
- missingTaskIds: string[];
189
- paired: PairedArmsComparison;
190
- relativeCost: Record<string, CrossSurfaceRelativeCost>;
191
- }
192
- interface CrossSurfacePairEvidence {
193
- bothTaskIds: string[];
194
- leftOnlyTaskIds: string[];
195
- rightOnlyTaskIds: string[];
196
- neitherTaskIds: string[];
197
- unobservedTaskIds: string[];
198
- }
199
- interface CrossSurfaceInteractionTask {
200
- taskId: string;
201
- /** Composition minus the additive expectation from the baseline and singles. */
202
- passInteraction: number | null;
203
- scoreInteraction: number | null;
204
- }
205
- interface CrossSurfaceInteractionEffect {
206
- perTask: CrossSurfaceInteractionTask[];
207
- n: number;
208
- nMissing: number;
209
- meanPassInteraction: number | null;
210
- meanScoreInteraction: number | null;
211
- passBootstrap: PairedBootstrapResult | null;
212
- scoreBootstrap: PairedBootstrapResult | null;
213
- }
214
- type CrossSurfacePairIncompatibilityReason = 'constituent_not_ready' | 'pair_incomplete' | 'baseline_regression' | 'interference' | 'no_incremental_resolution' | 'firing_below_minimum' | 'firing_unobserved' | 'effect_below_minimum' | 'effect_unobserved' | 'cost_limit_exceeded';
215
- interface CrossSurfacePairCompatibility {
216
- compatible: boolean;
217
- reasons: CrossSurfacePairIncompatibilityReason[];
218
- betterSingleCandidateId: string;
219
- }
220
- interface CrossSurfacePairwiseEntry {
221
- componentIds: [string, string];
222
- singleCandidateIds: [string, string];
223
- compositionCandidateId: string;
224
- benefitTaskIds: string[];
225
- regressionTaskIds: string[];
226
- synergyTaskIds: string[];
227
- interferenceTaskIds: string[];
228
- incrementalVsConstituents: [CrossSurfaceCandidateComparison, CrossSurfaceCandidateComparison];
229
- relativeCostToBaseline: Record<string, CrossSurfaceRelativeCost>;
230
- firing: CrossSurfacePairEvidence;
231
- effect: CrossSurfacePairEvidence;
232
- interaction: CrossSurfaceInteractionEffect;
233
- compatibility: CrossSurfacePairCompatibility;
234
- }
235
- interface CrossSurfaceRankedSingle {
236
- rank: number;
237
- candidateId: string;
238
- componentId: string;
239
- }
240
- interface CrossSurfaceBestSingleSelection {
241
- candidateId: string;
242
- componentId: string;
243
- ranking: CrossSurfaceRankedSingle[];
244
- }
245
- interface CrossSurfaceNaiveStackSelection {
246
- /** Every individually eligible single, stacked in canonical component order. */
247
- candidateId: string;
248
- componentIds: string[];
249
- }
250
- type CrossSurfaceAdditionRejectionReason = 'pair_incompatible' | 'full_bundle_not_evaluated' | 'bundle_incomplete' | 'baseline_regression' | 'no_incremental_resolution' | 'incremental_regression' | 'firing_below_minimum' | 'firing_unobserved' | 'effect_below_minimum' | 'effect_unobserved' | 'cost_limit_exceeded';
251
- interface CrossSurfaceAdditionDecision {
252
- additionCandidateId: string;
253
- additionComponentId: string;
254
- bundleCandidateId: string | null;
255
- incrementalResolutionTaskIds: string[];
256
- incrementalRegressionTaskIds: string[];
257
- incrementalMedianCost: Record<string, number> | null;
258
- eligible: boolean;
259
- selected: boolean;
260
- reasons: CrossSurfaceAdditionRejectionReason[];
261
- }
262
- interface CrossSurfaceCompositionStep {
263
- fromCandidateId: string;
264
- retainedComponentIds: string[];
265
- considered: CrossSurfaceAdditionDecision[];
266
- selectedCandidateId: string | null;
267
- }
268
- /** One deterministic growth path starting from a compatible two-surface seed. */
269
- interface CrossSurfaceInteractionPath {
270
- seedCandidateId: string;
271
- terminalCandidateId: string;
272
- terminalComponentIds: string[];
273
- qualified: boolean;
274
- steps: CrossSurfaceCompositionStep[];
275
- }
276
- interface CrossSurfaceInteractionAwareSelection {
277
- /** Compatible pair that seeded the selected deterministic growth path. */
278
- seedCandidateId: string;
279
- /** Candidate reached by the winning path, even if the minimum size is not met. */
280
- terminalCandidateId: string;
281
- terminalComponentIds: string[];
282
- /** null when no path produced a qualifying multi-component bundle. */
283
- selectedCandidateId: string | null;
284
- qualified: boolean;
285
- /** Every compatible pair seed is retained so seed choice cannot hide an interaction. */
286
- evaluatedPaths: CrossSurfaceInteractionPath[];
287
- /** Convenience alias for the winning path's steps. */
288
- steps: CrossSurfaceCompositionStep[];
289
- }
290
- interface CrossSurfaceSelections {
291
- bestSingle: CrossSurfaceBestSingleSelection | null;
292
- naiveStack: CrossSurfaceNaiveStackSelection | null;
293
- interactionAware: CrossSurfaceInteractionAwareSelection | null;
294
- }
295
- interface CrossSurfaceInteractionReport<TRow extends CrossSurfaceTaskRow = CrossSurfaceTaskRow> {
296
- taskIds: string[];
297
- componentIds: string[];
298
- candidateIds: string[];
299
- costMetrics: string[];
300
- /** Canonical candidate × task order; no input row is dropped. */
301
- rows: TRow[];
302
- missingAttempts: TRow[];
303
- invalidAttempts: TRow[];
304
- candidates: CrossSurfaceCandidateSummary[];
305
- pairwise: CrossSurfacePairwiseEntry[];
306
- selections: CrossSurfaceSelections;
307
- }
308
- //#endregion
309
- //#region src/campaign/cross-surface-interaction.d.ts
310
- /**
311
- * Build the complete cross-surface evidence matrix and derive all three frozen
312
- * candidates. The task/candidate/component orders are part of the input so
313
- * neither insertion order nor an after-the-fact tie-break can change a result.
314
- */
315
- declare function analyzeCrossSurfaceInteractions<TRow extends CrossSurfaceTaskRow>(input: AnalyzeCrossSurfaceInteractionsInput<TRow>): CrossSurfaceInteractionReport<TRow>;
316
- //#endregion
317
- //#region src/campaign/fixtures.d.ts
318
- type EvalFixtureValidationMode = 'vitest' | 'none';
319
- interface EvalFixtureFile {
320
- path: string;
321
- sha256: string;
322
- bytes: number;
323
- }
324
- interface EvalFixture {
325
- name: string;
326
- path: string;
327
- promptPath: string;
328
- evalPath?: string;
329
- packageJsonPath?: string;
330
- prompt: string;
331
- files: EvalFixtureFile[];
332
- fingerprint: string;
333
- }
334
- interface EvalFixtureScenario extends Scenario {
335
- kind: 'eval-fixture';
336
- fixtureName: string;
337
- fixturePath: string;
338
- promptPath: string;
339
- evalPath?: string;
340
- packageJsonPath?: string;
341
- prompt: string;
342
- fingerprint: string;
343
- }
344
- interface EvalFixtureLoadOptions {
345
- /** `vitest` requires EVAL.ts/EVAL.tsx and package.json type=module. `none` only requires PROMPT.md. */
346
- validation?: EvalFixtureValidationMode;
347
- /** Extra caller-owned knobs that affect fixture behavior, folded into the fingerprint. */
348
- fingerprintConfig?: unknown;
349
- }
350
- interface LoadEvalFixtureScenariosOptions extends EvalFixtureLoadOptions {
351
- names?: string[];
352
- }
353
- interface PlanEvalFixtureRunOptions<TArtifact = unknown> extends Pick<PlanCampaignRunOptions<EvalFixtureScenario, TArtifact>, 'dispatchRef' | 'judges' | 'seed' | 'reps' | 'resumable' | 'runDir'> {
354
- evalsDir: string;
355
- validation?: EvalFixtureValidationMode;
356
- fingerprintConfig?: unknown;
357
- names?: string[];
358
- storage?: CampaignStorage;
359
- }
360
- type EvalFixtureRunPlan = CampaignRunPlan & {
361
- fixtures: Array<Pick<EvalFixtureScenario, 'fixtureName' | 'fixturePath' | 'fingerprint'>>;
362
- };
363
- /** Walk `evalsDir` and return the relative name of every fixture directory (one containing an exact-case `PROMPT.md`). */
364
- declare function discoverEvalFixtures(evalsDir: string): string[];
365
- /**
366
- * Load ONE fixture by name: reads `PROMPT.md` (plus `EVAL.ts`/`EVAL.tsx` and `package.json` under
367
- * `vitest` validation) and content-fingerprints the full file set for cache identity.
368
- */
369
- declare function loadEvalFixture(evalsDir: string, name: string, options?: EvalFixtureLoadOptions): EvalFixture;
370
- /** Load fixtures (all discovered, or just `names`) as campaign `Scenario`s tagged `eval-fixture`. */
371
- declare function loadEvalFixtureScenarios(evalsDir: string, options?: LoadEvalFixtureScenariosOptions): EvalFixtureScenario[];
372
- /**
373
- * Dry-run planner for a fixture campaign: loads the scenarios, delegates to `planCampaignRun`,
374
- * and returns the plan plus each fixture's name/path/fingerprint.
375
- */
376
- declare function planEvalFixtureRun<TArtifact = unknown>(options: PlanEvalFixtureRunOptions<TArtifact>): EvalFixtureRunPlan;
377
- //#endregion
378
- //#region src/campaign/gates/neutralization-gate.d.ts
379
- interface NeutralizationGateOptions<TScenario extends Scenario = Scenario> {
380
- scenarios: TScenario[];
381
- /** Reject when the neutralized (content-blanked, footprint-matched) variant
382
- * reproduces at least this fraction of the candidate's held-out lift. Default
383
- * 0.5 — if blanking the content keeps half the lift, the content is decorative.
384
- * Equality rejects: a neutralized lift == threshold·candidateLift is decorative. */
385
- maxDecorativeFraction?: number;
386
- }
387
- /**
388
- * Composable placebo gate: ships only when the candidate's held-out lift is NOT
389
- * mostly reproduced by a footprint-matched neutralized variant.
390
- */
391
- declare function neutralizationGate<TArtifact, TScenario extends Scenario>(options: NeutralizationGateOptions<TScenario>): Gate<TArtifact, TScenario>;
392
- //#endregion
393
- //#region src/campaign/grounded-reflection.d.ts
394
- /**
395
- * Evidence grounding for reflective optimizers (GEPA-style revise loops).
396
- *
397
- * Two failure modes recur when an LLM revises an artifact from raw rollout
398
- * traces (first measured in agent-lab R358, where naive reflection REGRESSED
399
- * the score 0.375 -> 0.125 before these helpers fixed it):
400
- *
401
- * 1. The environment often hides WHY a rollout failed - a tool call can
402
- * succeed while an invisible downstream check fails - so the reviser
403
- * cannot see the cause in the transcript. The only reliable signal is the
404
- * field-level difference between what passing and failing rollouts did.
405
- * `rolloutArgumentDiff` computes that difference deterministically so the
406
- * reviser is handed the diff instead of being trusted to derive it.
407
- *
408
- * 2. Revisers invent plausible-but-wrong literal values ("use 'new'",
409
- * "use 'sent'") that no passing rollout ever used, turning every rollout
410
- * into a failure. `classifyUngroundedLiterals` mechanically detects them,
411
- * separating HARMFUL literals (ones failing rollouts actually used -
412
- * proven damage) from benign illustrations (e.g. a name example like
413
- * 'Doe'), so callers can hard-reject the former and merely log the latter.
414
- * Rejecting every ungrounded quoted word is too blunt: it killed a run
415
- * over a surname illustration before the severity split existed.
416
- *
417
- * Pure data in, data out: no LLM calls, no filesystem, no domain knowledge.
418
- */
419
- /** One tool/action call observed in a rollout: a name plus its arguments. */
420
- interface RolloutCall {
421
- readonly name: string;
422
- readonly args: Readonly<Record<string, unknown>>;
423
- }
424
- /** A scored rollout: its calls plus the scalar outcome used to split pass/fail. */
425
- interface ScoredRollout {
426
- /** Caller-meaningful identifier (task id, cell id) used only for reporting. */
427
- readonly id: string;
428
- /** Scalar outcome in [0, 1]; `passThreshold` splits passing from failing. */
429
- readonly score: number;
430
- readonly calls: readonly RolloutCall[];
431
- }
432
- interface RolloutArgumentDiffOptions {
433
- /** Rollouts with `score >= passThreshold` count as passing. Default 1. */
434
- readonly passThreshold?: number;
435
- /** Max distinct values listed per field per side in the rendered text. Default 4. */
436
- readonly maxValuesPerField?: number;
437
- }
438
- interface RolloutArgumentDiff {
439
- /** Human/LLM-readable per-field diff, one line per field. */
440
- readonly text: string;
441
- /** Lowercased stringified argument values seen in passing rollouts. */
442
- readonly passingValues: ReadonlySet<string>;
443
- /** Lowercased stringified argument values seen in failing rollouts. */
444
- readonly failingValues: ReadonlySet<string>;
445
- }
446
- /**
447
- * Deterministic per-field diff of call arguments between passing and failing
448
- * rollouts. A field set by failing rollouts but left unset by passing ones is
449
- * the classic poison-input signature; a field whose values differ across the
450
- * split points at the correct value. Feed `text` to the reviser verbatim.
451
- */
452
- declare function rolloutArgumentDiff(rollouts: readonly ScoredRollout[], opts?: RolloutArgumentDiffOptions): RolloutArgumentDiff;
453
- interface UngroundedLiteralReport {
454
- /** Quoted single-word literals in the text that no passing rollout used. */
455
- readonly ungrounded: readonly string[];
456
- /** The subset failing rollouts actually used - prescribing these is proven harmful. */
457
- readonly harmful: readonly string[];
458
- }
459
- /**
460
- * Scan revised artifact text for single-quoted single-word literals (the
461
- * "use exactly 'new'" pattern) that appear in no passing rollout's argument
462
- * values. Multi-word quotes pass (they are prose, not prescriptions).
463
- * Callers should reject on `harmful` (with a bounded retry) and at most log
464
- * `ungrounded` - see the module header for why the severities differ.
465
- */
466
- declare function classifyUngroundedLiterals(text: string, diff: Pick<RolloutArgumentDiff, 'passingValues' | 'failingValues'>): UngroundedLiteralReport;
467
- //#endregion
468
- //#region src/campaign/labeled-store/fs-adapter.d.ts
469
- interface FsLabeledScenarioStoreOptions {
470
- /** Root directory for JSONL files. Created if missing. */
471
- root: string;
472
- /** Per-source rate limit. When set, writes exceeding the cap are rejected
473
- * with a typed error. Default: no limit. */
474
- maxWritesPerMinutePerBucket?: number;
475
- /** Test seam — override `Date.now()` for deterministic tests. */
476
- now?: () => number;
477
- }
478
- /** Typed rejection from a labeled-scenario store (bad provenance, rate limit, invalid sample args) — carries a stable string `code`. */
479
- declare class LabeledScenarioStoreError extends Error {
480
- readonly code: string;
481
- constructor(code: string, message: string);
482
- }
483
- /**
484
- * Filesystem `LabeledScenarioStore`: appends one JSONL file per source with provenance and
485
- * rate-limit guards. For tests, local dev, and small workloads — high-throughput lands in Turso.
486
- */
487
- declare class FsLabeledScenarioStore implements LabeledScenarioStore {
488
- private readonly options;
489
- private readonly now;
490
- private readonly rateLimits;
491
- constructor(options: FsLabeledScenarioStoreOptions);
492
- observe(write: LabeledScenarioWrite): Promise<void>;
493
- sample(args: LabeledScenarioSampleArgs): Promise<LabeledScenarioRecord[]>;
494
- size(): Promise<{
495
- train: number;
496
- test: number;
497
- bySource: Record<string, number>;
498
- byTrust: Record<LabelTrust, number>;
499
- }>;
500
- private assertProvenance;
501
- private assertRateLimit;
502
- private toRecord;
503
- private pathForSource;
504
- }
505
- //#endregion
506
- //#region src/campaign/neutralize.d.ts
507
- /**
508
- * @module
509
- * Footprint-matched neutralization — the placebo control for content-vs-footprint
510
- * attribution in a promotion gate.
511
- *
512
- * A promoted surface can raise a held-out score two different ways:
513
- * 1. its CONTENT is informative (the thing we want to promote), or
514
- * 2. it merely added prompt/mount FOOTPRINT — more bytes, more lines, a longer
515
- * more authoritative-looking prompt — that the model spends attention on
516
- * regardless of what the bytes say.
517
- *
518
- * A held-out gate proves the candidate beat baseline; it cannot separate (1) from
519
- * (2). `neutralizeText` produces a variant that keeps the input's layout and
520
- * length while carrying ZERO information, so scoring it isolates the footprint
521
- * contribution (2). Feed the neutralized variant's scores to `neutralizationGate`:
522
- * any lift it still holds over baseline is decorative, and a candidate whose lift
523
- * survives neutralization is rejected however large its raw lift.
524
- */
525
- /**
526
- * Blank every non-whitespace character to a 1-byte filler while preserving all
527
- * whitespace. Line count, indentation, and word/line lengths are unchanged — so
528
- * the neutralized variant has the same layout and (for ASCII) the same byte
529
- * footprint as the input, but no readable content. Whitespace is preserved
530
- * deliberately: collapsing it would change the token structure and stop the
531
- * variant from being a true footprint match.
532
- */
533
- declare function neutralizeText(content: string): string;
534
- //#endregion
535
- //#region src/campaign/presets/run-profile-matrix.d.ts
536
- /** Thrown when the matrix is misconfigured (no profiles, missing resolved model evidence,
537
- * etc.). Distinct from `BackendIntegrityError`,
538
- * which signals a stub backend at run time. */
539
- declare class ProfileMatrixError extends AgentEvalError {
540
- constructor(message: string);
541
- }
542
- /** Dispatch for one cell: render `profile` against `scenario`, returning the
543
- * artifact the judges score. Run LLM work through `ctx.cost.runPaidCall` —
544
- * the integrity check depends on its receipt. */
545
- type ProfileDispatchFn<TScenario extends Scenario, TArtifact> = (profile: AgentProfile$1, scenario: TScenario, ctx: DispatchContext) => Promise<TArtifact>;
546
- interface RunProfileMatrixOptions<TScenario extends Scenario, TArtifact> {
547
- /** Axis 3 — the agent-under-test configurations. Each is one column. */
548
- profiles: AgentProfile$1[];
549
- /** Axis 1 — the persona/scenario corpus, run against every profile. */
550
- scenarios: TScenario[];
551
- /** Renders one (profile, scenario) cell. */
552
- dispatch: ProfileDispatchFn<TScenario, TArtifact>;
553
- /** The scoring axis. */
554
- judges?: JudgeConfig<TArtifact, TScenario>[];
555
- /** Where each profile's campaign writes artifacts/traces. One subdir per
556
- * profile. */
557
- runDir: string;
558
- /** Git SHA the harness ran from — stamped onto every RunRecord (mandatory
559
- * for paper-grade records). */
560
- commitSha: string;
561
- /** Additional stable identity for dispatch behavior that can change without
562
- * changing `commitSha`, such as a caller-owned executable or remote config. */
563
- dispatchRef?: string;
564
- /** Logical experiment id shared across the whole matrix so the promotion
565
- * gate can pair profiles on matched scenarios. Default: a hash of the
566
- * profile + scenario ids. */
567
- experimentId?: string;
568
- /** Which split these runs belong to. Default `'search'`. */
569
- splitTag?: RunSplitTag;
570
- /** Replicates per (profile, scenario) cell for CI bands. Default 1. */
571
- reps?: number;
572
- /** Campaign seed (per profile). Default 42. */
573
- seed?: number;
574
- /**
575
- * Backend-integrity posture, enforced AFTER the matrix completes:
576
- * - `'assert'` (default) — throw `BackendIntegrityError` if the run was a
577
- * stub (and, with `allowMixed:false`, if it was mixed).
578
- * - `'warn'` — log the verdict but never throw.
579
- * - `'off'` — skip the guard entirely (only for offline/replay analysis).
580
- */
581
- integrity?: 'assert' | 'warn' | 'off';
582
- /** Forwarded to `assertRealBackend`. Default true (tolerate partial 429
583
- * cascades); set false for strict CI gates. */
584
- allowMixed?: boolean;
585
- /** Max concurrent cells WITHIN each profile's campaign. Default 2. */
586
- maxConcurrency?: number;
587
- /** Max profile campaigns in flight. Default 1. Each profile keeps its own
588
- * run directory and cost ceiling; raise this when those resources are independent. */
589
- maxProfileConcurrency?: number;
590
- /** Cumulative USD cap per profile campaign. */
591
- costCeiling?: number;
592
- /** Capture flywheel — forwarded to each campaign. */
593
- labeledStore?: LabeledScenarioStore | 'off';
594
- captureSource?: LabeledScenarioSource;
595
- /** Storage backend. Default `fsCampaignStorage`. Pass
596
- * `inMemoryCampaignStorage()` for edge/CF-Worker/test runs. */
597
- storage?: CampaignStorage;
598
- /** Test seam — override the wall clock. */
599
- now?: () => Date;
600
- /** Optional persona key per scenario — drives the `byPersona` pivot. When
601
- * unset, `byPersona` is omitted. */
602
- personaOf?: (scenario: TScenario) => string;
603
- /** Validate every produced RunRecord with `validateRunRecord` (fail-loud).
604
- * Default true — catches bad model snapshots and non-finite judge dims at
605
- * the boundary instead of letting them poison downstream analysis. */
606
- validate?: boolean;
607
- /** Corpus-by-default: derive the trajectory text (`prompt` + `completion`)
608
- * for each cell from its artifact + scenario. When set, every produced
609
- * record carries `prompt`/`completion` (a `CorpusRecord`) so the run's
610
- * graded trajectories can be appended to the durable RL corpus with no
611
- * side-channel — `appendToCorpus(result.records, path)`. Fail-soft: a
612
- * throwing or undefined-returning extractor just omits the text. */
613
- corpusText?: (artifact: TArtifact, scenario: TScenario) => {
614
- prompt: string;
615
- completion: string;
616
- } | undefined;
617
- /**
618
- * Optional explicit row selection. The matrix identity remains based on the
619
- * complete profiles × scenarios × reps design; this invocation executes only
620
- * rows accepted by the predicate.
621
- */
622
- rowFilter?: (input: {
623
- profile: AgentProfile$1;
624
- scenario: TScenario;
625
- rep: number;
626
- }) => boolean;
627
- /** Stable matrix identity supplied by a persisted profile-matrix plan. */
628
- matrixId?: string;
629
- /** Reuse cached failed cells. Normal matrix runs retry them by default. */
630
- reuseFailedCells?: boolean;
631
- }
632
- interface ProfileSummary {
633
- profileId: string;
634
- profileHash: string;
635
- model: string;
636
- /** RunRecords produced for this profile (= scenarios × reps). */
637
- records: number;
638
- /** Mean across scored records, or null when the profile has no task labels. */
639
- meanComposite: number | null;
640
- /** Total cost, or null when any call's cost was not captured. */
641
- totalCostUsd: number | null;
642
- costProvenance: CostProvenance;
643
- /** Per-profile integrity verdict — surfaces a single profile that ran stub
644
- * even when the matrix as a whole looks real. */
645
- integrity: BackendIntegrityReport;
646
- }
647
- interface ScenarioRollup {
648
- meanComposite: number;
649
- n: number;
650
- }
651
- interface RunProfileMatrixResult<TArtifact, TScenario extends Scenario> {
652
- matrixId: string;
653
- experimentId: string;
654
- /** One RunRecord per (profile, scenario, rep) cell — the integrity-checked,
655
- * paper-grade output. Feed straight into `analyzeRuns`, `HeldOutGate`,
656
- * scorecards, the hosted wire format. */
657
- records: RunRecord[];
658
- byProfile: Record<string, ProfileSummary>;
659
- byScenario: Record<string, ScenarioRollup>;
660
- /** Present only when `personaOf` was supplied. */
661
- byPersona?: Record<string, ScenarioRollup>;
662
- /** Whole-matrix integrity report (the one `integrity:'assert'` enforces). */
663
- integrity: BackendIntegrityReport;
664
- /** The raw per-profile campaign results, keyed by profile id. */
665
- campaigns: Record<string, CampaignResult<TArtifact, TScenario>>;
666
- }
667
- /**
668
- * Profile × scenario matrix runner: fan N agent profiles across M scenarios, project each cell to a validated `RunRecord` with real token usage, and enforce the backend-integrity guard before returning.
669
- */
670
- declare function runProfileMatrix<TScenario extends Scenario, TArtifact>(opts: RunProfileMatrixOptions<TScenario, TArtifact>): Promise<RunProfileMatrixResult<TArtifact, TScenario>>;
671
- //#endregion
672
- //#region src/campaign/presets/playback.d.ts
673
- /** One step of a user story — what the user does. The driver interprets
674
- * `payload` (a Playwright selector + action, or a sandbox chat turn). */
675
- interface PlaybackStep {
676
- /** Human-readable action, captured verbatim in the UX narrative. */
677
- action: string;
678
- /** Driver-specific payload (e.g. `{ selector, fill }` or `{ turn }`). */
679
- payload?: Record<string, unknown>;
680
- }
681
- /**
682
- * A user story = a runnable product journey plus the requirements that define
683
- * "this story works". Each requirement is one Jira ticket line. Extends
684
- * `Scenario` so a catalog drops straight into `runProfileMatrix({ scenarios })`.
685
- */
686
- interface UserStory extends Scenario {
687
- /** Human-readable story title (the ticket headline). */
688
- title: string;
689
- /** Ordered steps the driver executes. */
690
- steps: PlaybackStep[];
691
- /** What must hold in the produced state for the story to pass. */
692
- requirements: CompletionRequirement[];
693
- }
694
- /** Dispatch context plus the profile under test (which cheap model, etc.). */
695
- interface PlaybackContext extends DispatchContext {
696
- profile: AgentProfile;
697
- }
698
- /**
699
- * Drives the real product through a story and returns the runtime event stream
700
- * `extractProducedState` consumes. Implemented by CONSUMERS —
701
- * `SandboxPlaybackDriver` (real API / sandbox workspace) and
702
- * `PlaywrightPlaybackDriver` (real UI) — because they depend on runtime /
703
- * browser infra the substrate must not import. The driver MUST report LLM
704
- * usage through `ctx.cost.runPaidCall` so the backend-integrity check sees real
705
- * tokens (a run that never reports tokens reads as a stub).
706
- */
707
- interface PlaybackDriver<TStory extends UserStory = UserStory> {
708
- run(story: TStory, ctx: PlaybackContext): Promise<readonly RuntimeEventLike[]>;
709
- }
710
- /**
711
- * Adapt a `PlaybackDriver` into a `runProfileMatrix` dispatch. The artifact the
712
- * matrix scores is the `ProducedState` extracted from the driver's event
713
- * stream — grade it with `scoreUserStory` (or a judge wrapping it).
714
- */
715
- declare function makePlaybackDispatch<TStory extends UserStory>(driver: PlaybackDriver<TStory>): ProfileDispatchFn<TStory, ProducedState>;
716
- /** A scored user story — the completion verdict plus its human title. */
717
- interface UserStoryVerdict extends CompletionVerdict {
718
- title: string;
719
- }
720
- /**
721
- * Score one story's produced state against its requirements. Thin wrapper over
722
- * `verifyCompletion` that builds the gold from the story and returns a
723
- * per-requirement PASS/FAIL verdict. `checkCorrectness` is injected — a
724
- * deterministic stub in tests, `createLlmCorrectnessChecker` in production.
725
- */
726
- declare function scoreUserStory(story: UserStory, state: ProducedState, checkCorrectness: CorrectnessChecker): Promise<UserStoryVerdict>;
727
- /** One row of the launch scoreboard — story × requirement → PASS/FAIL. */
728
- interface ScoreboardRow {
729
- storyId: string;
730
- storyTitle: string;
731
- reqId: string;
732
- reqTitle: string;
733
- status: 'PASS' | 'FAIL';
734
- evidence: string[];
735
- }
736
- /**
737
- * Flatten story verdicts into the per-requirement scoreboard — the literal
738
- * Jira tick-off: one row per (story, requirement) with PASS/FAIL and the
739
- * evidence behind the verdict.
740
- */
741
- declare function userStoryScoreboard(verdicts: readonly UserStoryVerdict[]): ScoreboardRow[];
742
- /** Launch-readiness headline counts rolled up from the per-requirement rows. */
743
- interface ScoreboardSummary {
744
- /** Distinct user stories on the board. */
745
- stories: number;
746
- /** Stories whose every requirement passed. */
747
- storiesFullyComplete: number;
748
- /** Total (story, requirement) rows. */
749
- requirements: number;
750
- /** Rows with status PASS. */
751
- passed: number;
752
- /** Rows with status FAIL. */
753
- failed: number;
754
- /** passed / requirements; 0 when there are no rows. */
755
- passRate: number;
756
- }
757
- /** Roll the per-requirement rows up into the launch headline counts. */
758
- declare function scoreboardSummary(rows: readonly ScoreboardRow[]): ScoreboardSummary;
759
- interface ScoreboardRenderOptions {
760
- /** Document H1. Defaults to a generic playback title. */
761
- title?: string;
762
- /** Key/value run metadata rendered under the headline (runId, backend, model, date). */
763
- meta?: Record<string, string>;
764
- /** Max chars of joined evidence shown per row. Default 160. */
765
- maxEvidenceChars?: number;
766
- }
767
- /**
768
- * Render the scoreboard as a launch-readiness Markdown document — the literal
769
- * "tick off every user story" artifact: a headline roll-up, the open tickets
770
- * (FAIL rows) up top as the launch blockers, then a per-story table of
771
- * requirement → PASS/FAIL with the evidence behind each verdict. Pure: same
772
- * rows in, same bytes out (no clock/random), so it is safe to snapshot.
773
- */
774
- declare function renderScoreboardMarkdown(rows: readonly ScoreboardRow[], opts?: ScoreboardRenderOptions): string;
775
- //#endregion
776
- //#region src/campaign/presets/segmented-profile-matrix.d.ts
777
- interface ProfileMatrixRow {
778
- rowId: string;
779
- ordinal: number;
780
- profileId: string;
781
- scenarioId: string;
782
- rep: number;
783
- }
784
- interface ProfileMatrixPlan<TScenario extends Scenario, TArtifact> {
785
- readonly schemaVersion: 1;
786
- readonly matrixId: string;
787
- readonly experimentId: string;
788
- readonly planDigest: `sha256:${string}`;
789
- readonly profiles: readonly AgentProfile$1[];
790
- readonly scenarios: readonly TScenario[];
791
- readonly judges: readonly JudgeConfig<TArtifact, TScenario>[];
792
- readonly reps: number;
793
- readonly seed: number;
794
- readonly splitTag: NonNullable<RunProfileMatrixOptions<TScenario, TArtifact>['splitTag']>;
795
- readonly commitSha: string;
796
- /** Stable dispatch implementation/configuration identity used by cell caches. */
797
- readonly dispatchRef: string;
798
- readonly integrity: NonNullable<RunProfileMatrixOptions<TScenario, TArtifact>['integrity']>;
799
- readonly allowMixed: boolean;
800
- readonly validate: boolean;
801
- readonly personaOf?: (scenario: TScenario) => string;
802
- readonly corpusText?: (artifact: TArtifact, scenario: TScenario) => {
803
- prompt: string;
804
- completion: string;
805
- } | undefined;
806
- readonly rows: readonly ProfileMatrixRow[];
807
- }
808
- interface CreateProfileMatrixPlanOptions<TScenario extends Scenario, TArtifact> extends Pick<RunProfileMatrixOptions<TScenario, TArtifact>, 'profiles' | 'scenarios' | 'judges' | 'commitSha' | 'experimentId' | 'splitTag' | 'reps' | 'seed' | 'integrity' | 'allowMixed' | 'validate' | 'personaOf' | 'corpusText'> {
809
- /** Stable dispatch implementation/configuration identity used by cell caches. */
810
- dispatchRef: string;
811
- }
812
- interface ProfileMatrixCoverage {
813
- expected: number;
814
- present: number;
815
- missing: string[];
816
- failed: string[];
817
- zeroScore: string[];
818
- }
819
- interface RunProfileMatrixSegmentOptions<TScenario extends Scenario, TArtifact> {
820
- plan: ProfileMatrixPlan<TScenario, TArtifact>;
821
- /** Stable external grant or attempt identity. Reuse it to resume. */
822
- segmentId: string;
823
- /** Explicit row ids from `plan.rows`; duplicate or unknown rows fail. */
824
- rows: readonly (string | ProfileMatrixRow)[];
825
- dispatch: ProfileDispatchFn<TScenario, TArtifact>;
826
- runDir: string;
827
- storage?: CampaignStorage;
828
- maxConcurrency?: number;
829
- maxProfileConcurrency?: number;
830
- costCeiling?: number;
831
- labeledStore?: RunProfileMatrixOptions<TScenario, TArtifact>['labeledStore'];
832
- captureSource?: RunProfileMatrixOptions<TScenario, TArtifact>['captureSource'];
833
- now?: () => Date;
834
- }
835
- interface ProfileMatrixSegmentResult<TArtifact, TScenario extends Scenario> {
836
- segmentId: string;
837
- rowIds: string[];
838
- matrix: RunProfileMatrixResult<TArtifact, TScenario>;
839
- coverage: ProfileMatrixCoverage;
840
- }
841
- interface FinalizeProfileMatrixOptions<TScenario extends Scenario, TArtifact> {
842
- plan: ProfileMatrixPlan<TScenario, TArtifact>;
843
- runDir: string;
844
- storage?: CampaignStorage;
845
- maxConcurrency?: number;
846
- maxProfileConcurrency?: number;
847
- now?: () => Date;
848
- }
849
- interface FinalizedProfileMatrixResult<TArtifact, TScenario extends Scenario> extends RunProfileMatrixResult<TArtifact, TScenario> {
850
- coverage: ProfileMatrixCoverage;
851
- }
852
- declare function createProfileMatrixPlan<TScenario extends Scenario, TArtifact>(opts: CreateProfileMatrixPlanOptions<TScenario, TArtifact>): ProfileMatrixPlan<TScenario, TArtifact>;
853
- declare function runProfileMatrixSegment<TScenario extends Scenario, TArtifact>(opts: RunProfileMatrixSegmentOptions<TScenario, TArtifact>): Promise<ProfileMatrixSegmentResult<TArtifact, TScenario>>;
854
- declare function finalizeProfileMatrix<TScenario extends Scenario, TArtifact>(opts: FinalizeProfileMatrixOptions<TScenario, TArtifact>): Promise<FinalizedProfileMatrixResult<TArtifact, TScenario>>;
855
- //#endregion
856
- //#region src/campaign/run-dir.d.ts
857
- /** The shared, out-of-repo root for campaign/benchmark run bundles. Keeping run
858
- * outputs here means they never land in a repo working tree (no per-repo
859
- * gitignore, no clutter, no accidental commits). Layout:
860
- * ~/.tangle/traces/<repo>/runs/<runName>/
861
- * where <repo> disambiguates runs across repos in one place. */
862
- declare function tangleTracesRoot(): string;
863
- /** Resolve a campaign `runDir`. An absolute path is honored as-is (the caller
864
- * chose an explicit location). A bare name is placed under the shared home root
865
- * so bundles never pollute a repo working tree — the default the harness should
866
- * compute so callers pass a *name*, not a path. */
867
- declare function resolveRunDir(runDir: string, repo?: string): string;
868
- //#endregion
869
- //#region src/campaign/scenario-selection.d.ts
870
- /**
871
- * Discriminative scenario selection (research claim E2).
872
- *
873
- * The OR benchmark is SATURATING: run 7 measured ~75% tied holdout cells — most
874
- * problems are solved optimally by the baseline AND every candidate, so those
875
- * paired cells carry zero signal. A random/balanced holdout split spends its
876
- * budget on scenarios that cannot separate candidates.
877
- *
878
- * This picks the holdout by DISCRIMINATION power instead: a scenario every
879
- * candidate scores identically (variance ~0) carries no signal; one where the
880
- * scores spread carries the most. We drop fully saturated ties so each paired
881
- * holdout cell is spent on a scenario that can actually move a verdict.
882
- */
883
- /** Per-scenario observation: the composite scores each candidate earned on it. */
884
- interface ScenarioSignal {
885
- scenarioId: string;
886
- /** Per-candidate composite scores observed for this scenario (>=1 values). */
887
- scores: number[];
888
- }
889
- interface DiscriminationScore {
890
- scenarioId: string;
891
- /** Higher = separates candidates more (spread of their scores). */
892
- discrimination: number;
893
- /** Higher = easier / more-saturated (mean of candidate scores). */
894
- meanScore: number;
895
- variance: number;
896
- /** variance ~0 AND meanScore at/above the ceiling ⇒ a saturated tie, no signal. */
897
- tied: boolean;
898
- }
899
- /**
900
- * Rank scenarios by how well they DISCRIMINATE candidates.
901
- *
902
- * `discrimination = variance` (spread of the candidate scores) — kept simple on
903
- * purpose; the headroom term (`saturationCeiling - meanScore`) only breaks ties
904
- * so that, among equally spread scenarios, the one with more room to improve
905
- * ranks first. Returned sorted by the deterministic order above.
906
- */
907
- declare function scoreDiscrimination(signals: ScenarioSignal[], opts?: {
908
- saturationCeiling?: number;
909
- }): DiscriminationScore[];
910
- /**
911
- * Select the top-`k` most discriminative scenario ids for a holdout, EXCLUDING
912
- * fully saturated ties when enough non-tied scenarios exist (a tie in the
913
- * holdout wastes a paired cell).
914
- *
915
- * Prefers non-tied scenarios; if fewer than `k` non-tied exist, fills with the
916
- * least-saturated tied ones (tied scenarios are already ordered least-saturated
917
- * first by `meanScore` asc). Deterministic. Throws if `k < 1`. If
918
- * `signals.length <= k`, returns all ids in discrimination order.
919
- */
920
- declare function selectDiscriminative(signals: ScenarioSignal[], k: number, opts?: {
921
- saturationCeiling?: number;
922
- }): string[];
923
- //#endregion
924
- //#region src/campaign/score-utils.d.ts
925
- /** Mean composite across cells with complete task-quality evidence.
926
- * Partial judge results remain on their cells but never enter this value.
927
- * A campaign with no complete score has no numeric mean and fails loudly. */
928
- declare function campaignMeanComposite<TArtifact, TScenario extends Scenario>(campaign: CampaignResult<TArtifact, TScenario>): number;
929
- /** Compare fixed-length lexicographic rank keys where each element is higher-is-better.
930
- * Returns a positive number when `a` ranks above `b`, negative when below, and
931
- * zero when equal. */
932
- declare function compareRankKeys(a: readonly number[], b: readonly number[]): number;
933
- interface CampaignBreakdown {
934
- /** Mean score per judge dimension across all cells. */
935
- dimensions: Record<string, number>;
936
- /** Per-scenario composite (mean over reps + judges) + the judge's free-form
937
- * `notes` for that scenario (the "why" a reflective proposer grounds on) +
938
- * an optional `emitted` excerpt of the candidate's raw output (the "what it
939
- * actually did" a reflective proposer grounds on). */
940
- scenarios: Array<{
941
- scenarioId: string;
942
- composite: number;
943
- notes?: string;
944
- emitted?: string;
945
- }>;
946
- }
947
- /** Per-candidate evidence a reflective/patch proposer grounds its next proposal
948
- * on: mean score per judge dimension + per-scenario composite. */
949
- declare function campaignBreakdown<TArtifact, TScenario extends Scenario>(campaign: CampaignResult<TArtifact, TScenario>): CampaignBreakdown;
950
- //#endregion
951
- //#region src/campaign/single-run-lock.d.ts
952
- /**
953
- * Single-run lock for evaluations that share one mutable environment.
954
- *
955
- * Two concurrent runs against a shared stateful gym silently corrupt each
956
- * other: each resets/mutates environment state mid-cell of the other, and
957
- * every score from both becomes garbage that LOOKS like worker variance
958
- * (agent-lab R357 burned hours on flip-flopping scores before tracing them
959
- * to exactly this). The fix is a pid lockfile: refuse to start while a live
960
- * holder exists, reclaim stale locks whose pid is gone, release only if the
961
- * lock is still ours.
962
- *
963
- * `alsoCheck` exists because independent runners can guard the same shared
964
- * resource with differently named lockfiles; a runner must respect all of
965
- * them even though it writes only its own.
966
- */
967
- interface SingleRunLockOptions {
968
- /** Lockfile this runner writes (and checks). */
969
- readonly lockPath: string;
970
- /** Other runners' lockfiles guarding the same resource; checked, never written. */
971
- readonly alsoCheck?: readonly string[];
972
- /** Install a process 'exit' hook that releases the lock. Default true. */
973
- readonly releaseOnExit?: boolean;
974
- /** Owner pid recorded in the lockfile metadata. Default process.pid. */
975
- readonly pid?: number;
976
- }
977
- interface SingleRunLock {
978
- /** Remove the lockfile if this process still owns it. Idempotent. */
979
- release(): void;
980
- }
981
- /**
982
- * Acquire the lock or throw naming the live holder. A stale lock (holder pid
983
- * no longer running) is reclaimed by one contender. An interrupted reclaim
984
- * leaves a marker that fails closed instead of admitting overlapping runs.
985
- */
986
- declare function acquireSingleRunLock(opts: SingleRunLockOptions): SingleRunLock;
987
- //#endregion
988
- //#region src/campaign/surface-identity.d.ts
989
- /** Validate the immutable identity shape; the owning executor verifies the Git objects and patch. */
990
- declare function assertCodeSurfaceIdentity(surface: unknown): asserts surface is CodeSurface;
991
- /**
992
- * Deterministic identity material for a component surface.
993
- *
994
- * `canonicalString` orders keys by UTF-16 code unit (RFC 8785), which is a
995
- * property of the value alone. The previous material ordered them with
996
- * `localeCompare`, which reads the host's collation — so the same surface
997
- * could produce two different identities on two machines, and the stored
998
- * identity would stop matching a recomputation of the identical surface.
999
- */
1000
- declare function componentSurfaceIdentityMaterial(surface: ComponentSurface): string;
1001
- /** Canonical, location-independent identity of a finalized code candidate.
1002
- * Commit metadata is excluded: two commits with the same base, final tree,
1003
- * and patch bytes are the same executable candidate. */
1004
- declare function codeSurfaceIdentityMaterial(surface: CodeSurface): string;
1005
- /** Full SHA-256 content identity for a prompt or finalized code surface. */
1006
- declare function surfaceContentHash(surface: MutableSurface): `sha256:${string}`;
1007
- /** Short loop key derived from the same content identity as provenance. */
1008
- declare function surfaceHash(surface: MutableSurface): string;
1009
- /** Canonical customer-visible description of the exact before/after surfaces. */
1010
- declare function renderSurfaceDiff(winnerSurface: MutableSurface, baselineSurface: MutableSurface): string;
1011
- //#endregion
1012
- //#region src/campaign/upstream-evaluators.d.ts
1013
- interface PhoenixEvaluationResultLike {
1014
- score?: number;
1015
- label?: string;
1016
- explanation?: string;
1017
- }
1018
- interface PhoenixEvaluatorLike<TRecord extends Record<string, unknown>> {
1019
- name: string;
1020
- kind: 'LLM' | 'CODE';
1021
- optimizationDirection?: 'MAXIMIZE' | 'MINIMIZE' | 'NEUTRAL';
1022
- evaluate(record: TRecord, context: UpstreamEvaluationContext): Promise<PhoenixEvaluationResultLike>;
1023
- }
1024
- interface AutoevalsScoreLike {
1025
- name: string;
1026
- score: number | null;
1027
- metadata?: Record<string, unknown>;
1028
- }
1029
- type AutoevalsScorerLike<TInput extends Record<string, unknown>> = (input: TInput, context: UpstreamEvaluationContext) => AutoevalsScoreLike | Promise<AutoevalsScoreLike>;
1030
- interface UpstreamEvaluationContext {
1031
- readonly signal: AbortSignal;
1032
- readonly callId?: string;
1033
- }
1034
- type PaidEvaluationOptions<TResult> = Pick<RunPaidCallInput<TResult>, 'maximumCharge' | 'receipt' | 'receiptFromError'> & {
1035
- model: string;
1036
- };
1037
- interface UpstreamJudgeOptions<TScenario extends Scenario> {
1038
- name?: string;
1039
- dimension?: string;
1040
- judgeVersion?: string;
1041
- appliesTo?: (scenario: TScenario) => boolean;
1042
- /** Convert the upstream score when its native scale is not higher-is-better. */
1043
- toComposite?: (score: number) => number;
1044
- }
1045
- declare function phoenixEvaluatorJudge<TRecord extends Record<string, unknown>, TArtifact, TScenario extends Scenario = Scenario>(evaluator: PhoenixEvaluatorLike<TRecord>, options: UpstreamJudgeOptions<TScenario> & {
1046
- mapInput(input: {
1047
- artifact: TArtifact;
1048
- scenario: TScenario;
1049
- }): TRecord;
1050
- paidCall?: PaidEvaluationOptions<PhoenixEvaluationResultLike>;
1051
- }): JudgeConfig<TArtifact, TScenario>;
1052
- declare function autoevalsScorerJudge<TInput extends Record<string, unknown>, TArtifact, TScenario extends Scenario = Scenario>(scorer: AutoevalsScorerLike<TInput>, options: UpstreamJudgeOptions<TScenario> & {
1053
- name: string;
1054
- mapInput(input: {
1055
- artifact: TArtifact;
1056
- scenario: TScenario;
1057
- }): TInput;
1058
- } & ({
1059
- kind: 'CODE';
1060
- paidCall?: never;
1061
- } | {
1062
- kind: 'LLM';
1063
- paidCall: PaidEvaluationOptions<AutoevalsScoreLike>;
1064
- })): JudgeConfig<TArtifact, TScenario>;
1065
- //#endregion
1066
- //#region src/campaign/worktree/index.d.ts
1067
- type GitOutput = string | Uint8Array;
1068
- type GitEnvironment = Readonly<Record<string, string>>;
1069
- type GitRunner = (args: string[], cwd: string, env?: GitEnvironment) => GitOutput;
1070
- interface Worktree {
1071
- /** Absolute path to the checked-out worktree directory. */
1072
- readonly path: string;
1073
- /** The branch the worktree is on (becomes the PR branch on promotion). */
1074
- readonly branch: string;
1075
- /** The ref the worktree was forked from. */
1076
- readonly baseRef: string;
1077
- /** Exact commit `baseRef` resolved to before the worktree was created. */
1078
- readonly baseCommit: string;
1079
- /** Exact tree object for `baseCommit`. */
1080
- readonly baseTree: string;
1081
- }
1082
- interface WorktreeAdapter {
1083
- /** Create an isolated worktree on a fresh branch off `baseRef`. */
1084
- create(opts: {
1085
- baseRef: string;
1086
- label: string;
1087
- }): Promise<Worktree>;
1088
- /** Commit pending changes, freeze the exact Git objects + binary patch, and
1089
- * verify the worktree still matches that identity. */
1090
- finalize(worktree: Worktree, summary: string): Promise<CodeSurface>;
1091
- /** Idempotently remove the worktree and branch. Safe to retry after partial cleanup. */
1092
- discard(worktree: Worktree): Promise<void>;
1093
- }
1094
- /** Typed failure from a `WorktreeAdapter` operation (create/finalize/discard) — wraps the underlying git error as `cause`. */
1095
- declare class WorktreeAdapterError extends Error {
1096
- readonly cause?: unknown;
1097
- constructor(message: string, cause?: unknown);
1098
- }
1099
- interface GitWorktreeAdapterOptions {
1100
- /** Repo root the worktrees fork from. */
1101
- repoRoot: string;
1102
- /** Directory worktrees are created under. Default: `<repoRoot>/.worktrees`. */
1103
- worktreeDir?: string;
1104
- /** Branch-name prefix. Default: `improve`. */
1105
- branchPrefix?: string;
1106
- /** Test seam — defaults to a real `git` runner. The return value must contain
1107
- * stdout verbatim, and runners that execute Git must forward the optional
1108
- * environment overrides used to isolate patch generation. */
1109
- git?: GitRunner;
1110
- }
1111
- interface CodeSurfaceVerification {
1112
- /** Verified worktree path. */
1113
- path: string;
1114
- /** Git's canonical root for the verified checkout. */
1115
- repoRoot: string;
1116
- /** Recomputed full content identity. */
1117
- contentHash: `sha256:${string}`;
1118
- /** Exact verified binary-patch bytes. Candidate-bundle builders encode this
1119
- * directly instead of reproducing Git diff options. */
1120
- patchBytes: Uint8Array;
1121
- }
1122
- /**
1123
- * Git-backed `WorktreeAdapter`: creates isolated worktrees on fresh branches, commits agent changes, and discards losers.
1124
- */
1125
- declare function gitWorktreeAdapter(opts: GitWorktreeAdapterOptions): WorktreeAdapter;
1126
- /** Verify a finalized code surface against its current checkout. This rejects
1127
- * dirty/ignored files, moved refs, missing Git objects, raw byte/mode
1128
- * mismatches, external symlinks, and submodules. */
1129
- declare function verifyCodeSurface(surface: CodeSurface, worktreeDir?: string): CodeSurfaceVerification;
1130
- /** Resolve a code candidate for evaluation only after verifying its immutable
1131
- * identity against the checkout at `worktreeRef`. */
1132
- declare function resolveWorktreePath(surface: CodeSurface, worktreeDir?: string): string;
1133
- //#endregion
1134
- export { makePlaybackDispatch as $, CrossSurfaceEvidenceBreakdown as $t, scoreDiscrimination as A, LoadEvalFixtureScenariosOptions as At, ProfileMatrixSegmentResult as B, CrossSurfaceAttemptCompleteness as Bt, acquireSingleRunLock as C, SearchLedgerIntegrityError as Cn, neutralizationGate as Ct, compareRankKeys as D, EvalFixtureRunPlan as Dt, campaignMeanComposite as E, EvalFixtureLoadOptions as Et, FinalizeProfileMatrixOptions as F, planEvalFixtureRun as Ft, PlaybackContext as G, CrossSurfaceCandidateEvidence as Gt, createProfileMatrixPlan as H, CrossSurfaceBootstrapPolicy as Ht, FinalizedProfileMatrixResult as I, analyzeCrossSurfaceInteractions as It, ScoreboardRenderOptions as J, CrossSurfaceComponent as Jt, PlaybackDriver as K, CrossSurfaceCandidateOutcome as Kt, ProfileMatrixCoverage as L, AnalyzeCrossSurfaceInteractionsInput as Lt, resolveRunDir as M, discoverEvalFixtures as Mt, tangleTracesRoot as N, loadEvalFixture as Nt, DiscriminationScore as O, EvalFixtureScenario as Ot, CreateProfileMatrixPlanOptions as P, loadEvalFixtureScenarios as Pt, UserStoryVerdict as Q, CrossSurfaceEligibility as Qt, ProfileMatrixPlan as R, CrossSurfaceAdditionDecision as Rt, SingleRunLockOptions as S, SearchLedgerError as Sn, NeutralizationGateOptions as St, campaignBreakdown as T, EvalFixtureFile as Tt, finalizeProfileMatrix as U, CrossSurfaceCandidate as Ut, RunProfileMatrixSegmentOptions as V, CrossSurfaceBestSingleSelection as Vt, runProfileMatrixSegment as W, CrossSurfaceCandidateComparison as Wt, ScoreboardSummary as X, CrossSurfaceCompositionStep as Xt, ScoreboardRow as Y, CrossSurfaceComponentEvidence as Yt, UserStory as Z, CrossSurfaceDistribution as Zt, componentSurfaceIdentityMaterial as _, TraceAnalystArtifact as _n, RolloutCall as _t, WorktreeAdapterError as a, CrossSurfaceInteractionTask as an, ProfileMatrixError as at, surfaceHash as b, traceAnalystQualityJudge as bn, classifyUngroundedLiterals as bt, verifyCodeSurface as c, CrossSurfacePairEvidence as cn, RunProfileMatrixResult as ct, PhoenixEvaluationResultLike as d, CrossSurfaceRankedSingle as dn, neutralizeText as dt, CrossSurfaceIneligibilityReason as en, renderScoreboardMarkdown as et, PhoenixEvaluatorLike as f, CrossSurfaceRelativeCost as fn, FsLabeledScenarioStore as ft, codeSurfaceIdentityMaterial as g, BuildTraceAnalystSurfaceDispatchOptions as gn, RolloutArgumentDiffOptions as gt, assertCodeSurfaceIdentity as h, CrossSurfaceTaskRow as hn, RolloutArgumentDiff as ht, WorktreeAdapter as i, CrossSurfaceInteractionReport as in, ProfileDispatchFn as it, selectDiscriminative as j, PlanEvalFixtureRunOptions as jt, ScenarioSignal as k, EvalFixtureValidationMode as kt, AutoevalsScoreLike as l, CrossSurfacePairIncompatibilityReason as ln, ScenarioRollup as lt, phoenixEvaluatorJudge as m, CrossSurfaceSelections as mn, LabeledScenarioStoreError as mt, GitWorktreeAdapterOptions as n, CrossSurfaceInteractionEffect as nn, scoreboardSummary as nt, gitWorktreeAdapter as o, CrossSurfaceNaiveStackSelection as on, ProfileSummary as ot, autoevalsScorerJudge as p, CrossSurfaceSelectionPolicy as pn, FsLabeledScenarioStoreOptions as pt, PlaybackStep as q, CrossSurfaceCandidateSummary as qt, Worktree as r, CrossSurfaceInteractionPath as rn, userStoryScoreboard as rt, resolveWorktreePath as s, CrossSurfacePairCompatibility as sn, RunProfileMatrixOptions as st, CodeSurfaceVerification as t, CrossSurfaceInteractionAwareSelection as tn, scoreUserStory as tt, AutoevalsScorerLike as u, CrossSurfacePairwiseEntry as un, runProfileMatrix as ut, renderSurfaceDiff as v, TraceAnalystScenario as vn, ScoredRollout as vt, CampaignBreakdown as w, EvalFixture as wt, SingleRunLock as x, SearchLedgerConflictError as xn, rolloutArgumentDiff as xt, surfaceContentHash as y, buildTraceAnalystSurfaceDispatch as yn, UngroundedLiteralReport as yt, ProfileMatrixRow as z, CrossSurfaceAdditionRejectionReason as zt };
1135
- //# sourceMappingURL=index-CFDffsKz.d.ts.map