@tangle-network/agent-eval 0.173.3 → 0.175.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +54 -0
- package/README.md +1 -1
- package/dist/{proposal-findings-bko3GGy-.js → abort-signal-CtzAM_sJ.js} +11 -11
- package/dist/abort-signal-CtzAM_sJ.js.map +1 -0
- package/dist/adapters/http.d.ts +2 -2
- package/dist/agent-profile-_xPxqVJt.d.ts +488 -0
- package/dist/agent-profile-_xPxqVJt.d.ts.map +1 -0
- package/dist/analyst/index.d.ts +7 -9
- package/dist/analyst/index.d.ts.map +1 -1
- package/dist/analyst/index.js +8 -8
- package/dist/{benchmark-C4wk_Sjr.js → benchmark-DQKzykkO.js} +2 -2
- package/dist/{benchmark-C4wk_Sjr.js.map → benchmark-DQKzykkO.js.map} +1 -1
- package/dist/{benchmark-command-BY9oscke.js → benchmark-command-D_5xG9LG.js} +13 -13
- package/dist/{benchmark-command-BY9oscke.js.map → benchmark-command-D_5xG9LG.js.map} +1 -1
- package/dist/benchmarks/index.d.ts +3 -4
- package/dist/benchmarks/index.d.ts.map +1 -1
- package/dist/benchmarks/index.js +3 -3
- package/dist/campaign/index.d.ts +5 -9
- package/dist/campaign/index.js +7 -7
- package/dist/{campaign-B3kPMU8S.js → campaign-BzMSCejE.js} +8 -8
- package/dist/{campaign-B3kPMU8S.js.map → campaign-BzMSCejE.js.map} +1 -1
- package/dist/{opencode-sqlite-eK6HW6dr.js → claude-jsonl-CxZZrDJ3.js} +9 -149
- package/dist/claude-jsonl-CxZZrDJ3.js.map +1 -0
- package/dist/cli.js +9 -2
- package/dist/cli.js.map +1 -1
- package/dist/{client-DlqdbM7n.d.ts → client-vyYQg3bm.d.ts} +2 -2
- package/dist/{client-DlqdbM7n.d.ts.map → client-vyYQg3bm.d.ts.map} +1 -1
- package/dist/contract/index.d.ts +9 -10
- package/dist/contract/index.js +8 -8
- package/dist/{default-registry-B0bKikCb.js → default-registry-DBqVI4pq.js} +5 -5
- package/dist/{default-registry-B0bKikCb.js.map → default-registry-DBqVI4pq.js.map} +1 -1
- package/dist/{default-registry-BKwc8bN5.d.ts → default-registry-FfNzaUHV.d.ts} +3 -3
- package/dist/{default-registry-BKwc8bN5.d.ts.map → default-registry-FfNzaUHV.d.ts.map} +1 -1
- package/dist/{define-agent-eval-CY6qdlGV.d.ts → define-agent-eval-V1jQyCDR.d.ts} +102 -11
- package/dist/define-agent-eval-V1jQyCDR.d.ts.map +1 -0
- package/dist/{define-agent-eval-8h3lXXee.js → define-agent-eval-ox5McL6e.js} +331 -144
- package/dist/define-agent-eval-ox5McL6e.js.map +1 -0
- package/dist/{dspy-rlm-engine-CF0t2ITD.js → dspy-rlm-engine-Caz2pl4L.js} +3 -3
- package/dist/{dspy-rlm-engine-CF0t2ITD.js.map → dspy-rlm-engine-Caz2pl4L.js.map} +1 -1
- package/dist/{engine-DhFir3Ys.d.ts → engine-CvW_I72-.d.ts} +2 -2
- package/dist/{engine-DhFir3Ys.d.ts.map → engine-CvW_I72-.d.ts.map} +1 -1
- package/dist/experiment/index.d.ts +1 -4
- package/dist/experiment/index.d.ts.map +1 -1
- package/dist/{external-optimizer-process-BwITA9Jp.js → external-optimizer-process-CxnFL1hd.js} +2 -2
- package/dist/{external-optimizer-process-BwITA9Jp.js.map → external-optimizer-process-CxnFL1hd.js.map} +1 -1
- package/dist/{external-optimizer-subprocess-wBWeoG6A.js → external-optimizer-subprocess-CQi27uEI.js} +2 -2
- package/dist/{external-optimizer-subprocess-wBWeoG6A.js.map → external-optimizer-subprocess-CQi27uEI.js.map} +1 -1
- package/dist/fuzz.js +1 -1
- package/dist/fuzz.js.map +1 -1
- package/dist/hosted/index.d.ts +1 -1
- package/dist/{index-D0Db5X-4.d.ts → index-BAAiSF3_.d.ts} +5 -5
- package/dist/{index-D0Db5X-4.d.ts.map → index-BAAiSF3_.d.ts.map} +1 -1
- package/dist/{index-BQqOjerE.d.ts → index-BTrx5s8m.d.ts} +8 -9
- package/dist/index-BTrx5s8m.d.ts.map +1 -0
- package/dist/index-DKXuBPXf.d.ts +3840 -0
- package/dist/index-DKXuBPXf.d.ts.map +1 -0
- package/dist/index.d.ts +11 -13
- package/dist/index.d.ts.map +1 -1
- package/dist/index.js +11 -11
- package/dist/{integrity-BWywb34E.js → integrity-DsHWCebQ.js} +11 -435
- package/dist/integrity-DsHWCebQ.js.map +1 -0
- package/dist/{kind-factory-gP6lDySe.js → kind-factory-BLvL-E44.js} +2 -2
- package/dist/{kind-factory-gP6lDySe.js.map → kind-factory-BLvL-E44.js.map} +1 -1
- package/dist/{llm-judge-BfqMFo4h.js → llm-judge-DmNaBrXB.js} +2541 -2435
- package/dist/llm-judge-DmNaBrXB.js.map +1 -0
- package/dist/{matrix-DGu8KhSs.d.ts → matrix-CJtXz1ky.d.ts} +2 -2
- package/dist/{matrix-DGu8KhSs.d.ts.map → matrix-CJtXz1ky.d.ts.map} +1 -1
- package/dist/multishot/golden/index.d.ts +1 -1
- package/dist/multishot/index.d.ts +2 -2
- package/dist/openapi.json +1 -1
- package/dist/opencode-sqlite-CNw3vubS.js +145 -0
- package/dist/opencode-sqlite-CNw3vubS.js.map +1 -0
- package/dist/{produced-state-D91uDvQw.js → produced-state-B8mw6zj9.js} +2 -2
- package/dist/{produced-state-D91uDvQw.js.map → produced-state-B8mw6zj9.js.map} +1 -1
- package/dist/report-command-DKlXfU5r.js +1528 -0
- package/dist/report-command-DKlXfU5r.js.map +1 -0
- package/dist/rl.d.ts +1 -1
- package/dist/rl.d.ts.map +1 -1
- package/dist/rl.js.map +1 -1
- package/dist/rollout/index.js +3 -2
- package/dist/{rollout-C-znbbYg.js → rollout-CGlDq1GI.js} +3 -2
- package/dist/{rollout-C-znbbYg.js.map → rollout-CGlDq1GI.js.map} +1 -1
- package/dist/{semantic-concept-judge-Dok7_35a.js → semantic-concept-judge-E3s_fEjB.js} +3 -3
- package/dist/{semantic-concept-judge-Dok7_35a.js.map → semantic-concept-judge-E3s_fEjB.js.map} +1 -1
- package/dist/{skillopt-optimization-method-DDw3v3gA.js → skillopt-optimization-method-f7399oGb.js} +5 -5
- package/dist/{skillopt-optimization-method-DDw3v3gA.js.map → skillopt-optimization-method-f7399oGb.js.map} +1 -1
- package/dist/statistical-heldout-Cqb73yE9.d.ts +1127 -0
- package/dist/statistical-heldout-Cqb73yE9.d.ts.map +1 -0
- package/dist/{store-otlp-Dow0pk_5.js → store-otlp-DV_H2HDu.js} +2 -2
- package/dist/{store-otlp-Dow0pk_5.js.map → store-otlp-DV_H2HDu.js.map} +1 -1
- package/dist/{store-tool-spans-CCZNsihA.d.ts → store-tool-spans-4o55ABER.d.ts} +3 -3
- package/dist/{store-tool-spans-CCZNsihA.d.ts.map → store-tool-spans-4o55ABER.d.ts.map} +1 -1
- package/dist/{store-tool-spans-CeNj_m2L.js → store-tool-spans-B9tjys_h.js} +3 -3
- package/dist/{store-tool-spans-CeNj_m2L.js.map → store-tool-spans-B9tjys_h.js.map} +1 -1
- package/dist/supervisor-run/index.d.ts +71 -6
- package/dist/supervisor-run/index.d.ts.map +1 -1
- package/dist/supervisor-run/index.js +6 -1357
- package/dist/supervisor-run/index.js.map +1 -1
- package/dist/{task-failure-attributes-CZjZeBsY.js → task-failure-attributes-CUy9mkIY.js} +2 -2
- package/dist/{task-failure-attributes-CZjZeBsY.js.map → task-failure-attributes-CUy9mkIY.js.map} +1 -1
- package/dist/terminal-record-Ce9_UjRz.js +539 -0
- package/dist/terminal-record-Ce9_UjRz.js.map +1 -0
- package/dist/{tool-groups-Cp4Xdzrp.d.ts → tool-groups-DAe1t6zb.d.ts} +2 -2
- package/dist/tool-groups-DAe1t6zb.d.ts.map +1 -0
- package/dist/trace-repair/index.d.ts +1 -1
- package/dist/traces.d.ts +2 -2
- package/dist/traces.js +4 -4
- package/dist/{types-Ba5UQyVD.d.ts → types-BJz2CPTM.d.ts} +2 -2
- package/dist/{types-Ba5UQyVD.d.ts.map → types-BJz2CPTM.d.ts.map} +1 -1
- package/dist/{types-CiWITkGo.js → types-DQ0e2E7y.js} +2 -2
- package/dist/types-DQ0e2E7y.js.map +1 -0
- package/dist/{types-BDV4PiMR.d.ts → types-Dd1ejaeI.d.ts} +2 -2
- package/dist/{types-BDV4PiMR.d.ts.map → types-Dd1ejaeI.d.ts.map} +1 -1
- package/dist/{types-CoPUTiXb.d.ts → types-vUdAx2Cj.d.ts} +65 -3
- package/dist/types-vUdAx2Cj.d.ts.map +1 -0
- package/docs/campaign-proposers.md +42 -0
- package/package.json +1 -1
- package/dist/agent-profile-B9_GGsG8.d.ts +0 -84
- package/dist/agent-profile-B9_GGsG8.d.ts.map +0 -1
- package/dist/backend-integrity-CeuTgqsd.d.ts +0 -280
- package/dist/backend-integrity-CeuTgqsd.d.ts.map +0 -1
- package/dist/benchmark-BjLGkfnN.d.ts +0 -236
- package/dist/benchmark-BjLGkfnN.d.ts.map +0 -1
- package/dist/define-agent-eval-8h3lXXee.js.map +0 -1
- package/dist/define-agent-eval-CY6qdlGV.d.ts.map +0 -1
- package/dist/external-optimizer-contracts-CQCpyrIL.d.ts +0 -172
- package/dist/external-optimizer-contracts-CQCpyrIL.d.ts.map +0 -1
- package/dist/heldout-gate-Df5hsqmm.d.ts +0 -453
- package/dist/heldout-gate-Df5hsqmm.d.ts.map +0 -1
- package/dist/index-BQqOjerE.d.ts.map +0 -1
- package/dist/index-CFDffsKz.d.ts +0 -1135
- package/dist/index-CFDffsKz.d.ts.map +0 -1
- package/dist/integrity-BWywb34E.js.map +0 -1
- package/dist/llm-judge-BfqMFo4h.js.map +0 -1
- package/dist/opencode-sqlite-eK6HW6dr.js.map +0 -1
- package/dist/power-preflight-Ptse_Kq7.d.ts +0 -117
- package/dist/power-preflight-Ptse_Kq7.d.ts.map +0 -1
- package/dist/pre-registration-BoI4ucR3.d.ts +0 -592
- package/dist/pre-registration-BoI4ucR3.d.ts.map +0 -1
- package/dist/promotion-policy-CvMda3kU.d.ts +0 -134
- package/dist/promotion-policy-CvMda3kU.d.ts.map +0 -1
- package/dist/proposal-findings-bko3GGy-.js.map +0 -1
- package/dist/provenance-CRY67X50.d.ts +0 -1995
- package/dist/provenance-CRY67X50.d.ts.map +0 -1
- package/dist/statistical-heldout-DTyB_6-1.d.ts +0 -295
- package/dist/statistical-heldout-DTyB_6-1.d.ts.map +0 -1
- package/dist/tool-groups-Cp4Xdzrp.d.ts.map +0 -1
- package/dist/types-CiWITkGo.js.map +0 -1
- package/dist/types-CoPUTiXb.d.ts.map +0 -1
package/dist/index-CFDffsKz.d.ts
DELETED
|
@@ -1,1135 +0,0 @@
|
|
|
1
|
-
import { c as ValidationError, t as AgentEvalError } from "./errors-DEE6u6ot.js";
|
|
2
|
-
import { T as RunPaidCallInput, p as CostProvenance } from "./cost-ledger-DbQdN3nO.js";
|
|
3
|
-
import { a as RunRecord, s as RunSplitTag } from "./run-record-DTv1MdjK.js";
|
|
4
|
-
import { f as AnalystUsageReceipt, i as AnalystFinding, l as AnalystRunResult, w as TraceAnalysisStore } from "./types-DN2WdT5S.js";
|
|
5
|
-
import { _ as PairedArmsComparison } from "./pre-registration-BoI4ucR3.js";
|
|
6
|
-
import { d as PairedBootstrapResult } from "./paired-promotion-decision-CGzg0cI_.js";
|
|
7
|
-
import { A as LabeledScenarioWrite, D as LabeledScenarioSampleArgs, E as LabeledScenarioRecord, O as LabeledScenarioSource, R as Scenario, S as JudgeConfig, T as LabelTrust, a as CampaignResult, d as DispatchContext, j as MutableSurface, k as LabeledScenarioStore, l as CodeSurface, p as Gate, u as ComponentSurface } from "./types-Ba5UQyVD.js";
|
|
8
|
-
import "./heldout-gate-Df5hsqmm.js";
|
|
9
|
-
import { In as CampaignRunPlan, Kn as CampaignStorage, Rn as PlanCampaignRunOptions } from "./provenance-CRY67X50.js";
|
|
10
|
-
import "./promotion-policy-CvMda3kU.js";
|
|
11
|
-
import { d as CompletionVerdict, f as CorrectnessChecker, h as ProducedState, n as BackendIntegrityReport, s as RuntimeEventLike, u as CompletionRequirement } from "./backend-integrity-CeuTgqsd.js";
|
|
12
|
-
import { _ as AnalystIssueExpectation, a as AnalystBenchmarkLabelState, t as AnalystBenchmarkCase } from "./benchmark-BjLGkfnN.js";
|
|
13
|
-
import { t as AgentProfile$1 } from "./agent-profile-B9_GGsG8.js";
|
|
14
|
-
import "./statistical-heldout-DTyB_6-1.js";
|
|
15
|
-
import { AgentProfile } from "@tangle-network/agent-interface";
|
|
16
|
-
//#region src/campaign/search-ledger-errors.d.ts
|
|
17
|
-
/** Base error for invalid search-ledger input or operations. */
|
|
18
|
-
declare class SearchLedgerError extends ValidationError {}
|
|
19
|
-
/** Error raised when durable search-ledger data fails an integrity check. */
|
|
20
|
-
declare class SearchLedgerIntegrityError extends SearchLedgerError {}
|
|
21
|
-
/** Error raised when an event identifier is reused with different content. */
|
|
22
|
-
declare class SearchLedgerConflictError extends SearchLedgerError {}
|
|
23
|
-
//#endregion
|
|
24
|
-
//#region src/campaign/analyst-surface.d.ts
|
|
25
|
-
interface TraceAnalystScenario extends Scenario {
|
|
26
|
-
kind: 'trace-analyst';
|
|
27
|
-
traceStore: TraceAnalysisStore;
|
|
28
|
-
labelState: AnalystBenchmarkLabelState;
|
|
29
|
-
expectedIssues: readonly AnalystIssueExpectation[];
|
|
30
|
-
labeledEvidence?: AnalystBenchmarkCase['labeledEvidence'];
|
|
31
|
-
}
|
|
32
|
-
interface TraceAnalystArtifact {
|
|
33
|
-
findings: readonly AnalystFinding[];
|
|
34
|
-
usage?: AnalystUsageReceipt;
|
|
35
|
-
run?: AnalystRunResult;
|
|
36
|
-
}
|
|
37
|
-
interface BuildTraceAnalystSurfaceDispatchOptions {
|
|
38
|
-
analyze(input: {
|
|
39
|
-
instructions: string;
|
|
40
|
-
traceStore: TraceAnalysisStore;
|
|
41
|
-
runId: string;
|
|
42
|
-
signal: AbortSignal;
|
|
43
|
-
}): Promise<TraceAnalystArtifact>;
|
|
44
|
-
}
|
|
45
|
-
declare function buildTraceAnalystSurfaceDispatch(options: BuildTraceAnalystSurfaceDispatchOptions): (surface: MutableSurface, scenario: TraceAnalystScenario, context: DispatchContext) => Promise<TraceAnalystArtifact>;
|
|
46
|
-
declare function traceAnalystQualityJudge(): JudgeConfig<TraceAnalystArtifact, TraceAnalystScenario>;
|
|
47
|
-
//#endregion
|
|
48
|
-
//#region src/campaign/cross-surface-types.d.ts
|
|
49
|
-
/** Whether one candidate attempt produced a usable executable outcome. */
|
|
50
|
-
type CrossSurfaceAttemptCompleteness = 'complete' | 'missing' | 'invalid';
|
|
51
|
-
/** One independently proposed change on one caller-defined surface. */
|
|
52
|
-
interface CrossSurfaceComponent {
|
|
53
|
-
componentId: string;
|
|
54
|
-
surfaceId: string;
|
|
55
|
-
/** Explicitly controls whether this component may anchor the best-single arm. */
|
|
56
|
-
bestSingleEligible: boolean;
|
|
57
|
-
}
|
|
58
|
-
/** Immutable identity for a single candidate or a materialized composition. */
|
|
59
|
-
interface CrossSurfaceCandidate {
|
|
60
|
-
candidateId: string;
|
|
61
|
-
componentIds: string[];
|
|
62
|
-
contentHash: string;
|
|
63
|
-
artifactBytes: number;
|
|
64
|
-
}
|
|
65
|
-
/** Per-component trace evidence captured during one task attempt. */
|
|
66
|
-
interface CrossSurfaceComponentEvidence {
|
|
67
|
-
componentId: string;
|
|
68
|
-
/** null means the trace could not establish whether the component fired. */
|
|
69
|
-
fired: boolean | null;
|
|
70
|
-
/** null means the trace could not establish whether the component changed behavior. */
|
|
71
|
-
effectObserved: boolean | null;
|
|
72
|
-
}
|
|
73
|
-
/**
|
|
74
|
-
* Canonical per-task input row. Consumers may extend this interface with
|
|
75
|
-
* receipt, trace, retry, or failure details; the report preserves the original
|
|
76
|
-
* row object rather than projecting those details away.
|
|
77
|
-
*/
|
|
78
|
-
interface CrossSurfaceTaskRow {
|
|
79
|
-
taskId: string;
|
|
80
|
-
candidateId: string;
|
|
81
|
-
/** Repeated here so every persisted row remains self-describing. */
|
|
82
|
-
componentIds: string[];
|
|
83
|
-
completeness: CrossSurfaceAttemptCompleteness;
|
|
84
|
-
pass: boolean | null;
|
|
85
|
-
score: number | null;
|
|
86
|
-
/**
|
|
87
|
-
* Per-attempt deployment measurements. Every declared metric must have a
|
|
88
|
-
* known, non-negative value. Proposal, analysis, and selection spend belongs
|
|
89
|
-
* in the search ledger rather than being spread across task cells.
|
|
90
|
-
*/
|
|
91
|
-
cost: Record<string, number | null>;
|
|
92
|
-
componentEvidence: CrossSurfaceComponentEvidence[];
|
|
93
|
-
/** Required for missing or invalid attempts; forbidden for complete attempts. */
|
|
94
|
-
rejectReason: string | null;
|
|
95
|
-
}
|
|
96
|
-
interface CrossSurfaceBootstrapPolicy {
|
|
97
|
-
seed: number;
|
|
98
|
-
resamples: number;
|
|
99
|
-
confidence: number;
|
|
100
|
-
}
|
|
101
|
-
/** Predeclared candidate eligibility and composition policy. */
|
|
102
|
-
interface CrossSurfaceSelectionPolicy {
|
|
103
|
-
minimumFiringTasks: number;
|
|
104
|
-
minimumEffectTasks: number;
|
|
105
|
-
requireObservedFiring: boolean;
|
|
106
|
-
requireObservedEffect: boolean;
|
|
107
|
-
/** Only named metrics are constrained; all declared metrics are still reported. */
|
|
108
|
-
maximumMedianCostRatioToBaseline: Record<string, number>;
|
|
109
|
-
/** A smaller terminal bundle is reported but cannot become the selected arm. */
|
|
110
|
-
minimumBundleComponents: number;
|
|
111
|
-
}
|
|
112
|
-
interface AnalyzeCrossSurfaceInteractionsInput<TRow extends CrossSurfaceTaskRow = CrossSurfaceTaskRow> {
|
|
113
|
-
components: readonly CrossSurfaceComponent[];
|
|
114
|
-
candidates: readonly CrossSurfaceCandidate[];
|
|
115
|
-
rows: readonly TRow[];
|
|
116
|
-
baselineCandidateId: string;
|
|
117
|
-
/** Exact shared task axis and its canonical output order. */
|
|
118
|
-
taskOrder: readonly string[];
|
|
119
|
-
/** Canonical materialization order for component sets and the naive stack. */
|
|
120
|
-
componentOrder: readonly string[];
|
|
121
|
-
/** Final deterministic tie-break; lower index wins. */
|
|
122
|
-
candidateOrder: readonly string[];
|
|
123
|
-
/** Declares every cost key and the order used for cost tie-breaks. */
|
|
124
|
-
costMetricOrder: readonly string[];
|
|
125
|
-
bootstrap: CrossSurfaceBootstrapPolicy;
|
|
126
|
-
selection: CrossSurfaceSelectionPolicy;
|
|
127
|
-
}
|
|
128
|
-
interface CrossSurfaceDistribution {
|
|
129
|
-
n: number;
|
|
130
|
-
min: number;
|
|
131
|
-
median: number;
|
|
132
|
-
mean: number;
|
|
133
|
-
max: number;
|
|
134
|
-
total: number;
|
|
135
|
-
}
|
|
136
|
-
interface CrossSurfaceEvidenceBreakdown {
|
|
137
|
-
componentId: string;
|
|
138
|
-
observedTaskIds: string[];
|
|
139
|
-
notObservedTaskIds: string[];
|
|
140
|
-
unobservedTaskIds: string[];
|
|
141
|
-
}
|
|
142
|
-
interface CrossSurfaceCandidateEvidence {
|
|
143
|
-
byComponent: CrossSurfaceEvidenceBreakdown[];
|
|
144
|
-
allObservedTaskIds: string[];
|
|
145
|
-
someObservedTaskIds: string[];
|
|
146
|
-
noneObservedTaskIds: string[];
|
|
147
|
-
unobservedTaskIds: string[];
|
|
148
|
-
}
|
|
149
|
-
type CrossSurfaceIneligibilityReason = 'missing_attempt' | 'invalid_attempt' | 'baseline_outcome_missing' | 'benefit_not_greater_than_regression' | 'firing_below_minimum' | 'firing_unobserved' | 'effect_below_minimum' | 'effect_unobserved' | 'cost_limit_exceeded';
|
|
150
|
-
interface CrossSurfaceEligibility {
|
|
151
|
-
eligible: boolean;
|
|
152
|
-
reasons: CrossSurfaceIneligibilityReason[];
|
|
153
|
-
}
|
|
154
|
-
interface CrossSurfaceCandidateOutcome {
|
|
155
|
-
resolvedTaskIds: string[];
|
|
156
|
-
failedTaskIds: string[];
|
|
157
|
-
missingTaskIds: string[];
|
|
158
|
-
invalidTaskIds: string[];
|
|
159
|
-
benefitTaskIds: string[];
|
|
160
|
-
regressionTaskIds: string[];
|
|
161
|
-
comparisonMissingTaskIds: string[];
|
|
162
|
-
netBenefit: number;
|
|
163
|
-
}
|
|
164
|
-
interface CrossSurfaceCandidateSummary {
|
|
165
|
-
candidate: CrossSurfaceCandidate;
|
|
166
|
-
outcome: CrossSurfaceCandidateOutcome;
|
|
167
|
-
score: CrossSurfaceDistribution | null;
|
|
168
|
-
costs: Record<string, CrossSurfaceDistribution>;
|
|
169
|
-
firing: CrossSurfaceCandidateEvidence;
|
|
170
|
-
effect: CrossSurfaceCandidateEvidence;
|
|
171
|
-
/** Reuses the package's paired McNemar/risk-difference/bootstrap statistics. */
|
|
172
|
-
comparisonToBaseline: PairedArmsComparison | null;
|
|
173
|
-
/** null only for the fixed baseline. */
|
|
174
|
-
eligibility: CrossSurfaceEligibility | null;
|
|
175
|
-
}
|
|
176
|
-
interface CrossSurfaceRelativeCost {
|
|
177
|
-
treatmentMedian: number;
|
|
178
|
-
comparatorMedian: number;
|
|
179
|
-
medianDelta: number;
|
|
180
|
-
/** null when the comparator median is zero but the treatment median is not. */
|
|
181
|
-
medianRatio: number | null;
|
|
182
|
-
}
|
|
183
|
-
interface CrossSurfaceCandidateComparison {
|
|
184
|
-
comparatorCandidateId: string;
|
|
185
|
-
treatmentCandidateId: string;
|
|
186
|
-
winsTaskIds: string[];
|
|
187
|
-
regressionTaskIds: string[];
|
|
188
|
-
missingTaskIds: string[];
|
|
189
|
-
paired: PairedArmsComparison;
|
|
190
|
-
relativeCost: Record<string, CrossSurfaceRelativeCost>;
|
|
191
|
-
}
|
|
192
|
-
interface CrossSurfacePairEvidence {
|
|
193
|
-
bothTaskIds: string[];
|
|
194
|
-
leftOnlyTaskIds: string[];
|
|
195
|
-
rightOnlyTaskIds: string[];
|
|
196
|
-
neitherTaskIds: string[];
|
|
197
|
-
unobservedTaskIds: string[];
|
|
198
|
-
}
|
|
199
|
-
interface CrossSurfaceInteractionTask {
|
|
200
|
-
taskId: string;
|
|
201
|
-
/** Composition minus the additive expectation from the baseline and singles. */
|
|
202
|
-
passInteraction: number | null;
|
|
203
|
-
scoreInteraction: number | null;
|
|
204
|
-
}
|
|
205
|
-
interface CrossSurfaceInteractionEffect {
|
|
206
|
-
perTask: CrossSurfaceInteractionTask[];
|
|
207
|
-
n: number;
|
|
208
|
-
nMissing: number;
|
|
209
|
-
meanPassInteraction: number | null;
|
|
210
|
-
meanScoreInteraction: number | null;
|
|
211
|
-
passBootstrap: PairedBootstrapResult | null;
|
|
212
|
-
scoreBootstrap: PairedBootstrapResult | null;
|
|
213
|
-
}
|
|
214
|
-
type CrossSurfacePairIncompatibilityReason = 'constituent_not_ready' | 'pair_incomplete' | 'baseline_regression' | 'interference' | 'no_incremental_resolution' | 'firing_below_minimum' | 'firing_unobserved' | 'effect_below_minimum' | 'effect_unobserved' | 'cost_limit_exceeded';
|
|
215
|
-
interface CrossSurfacePairCompatibility {
|
|
216
|
-
compatible: boolean;
|
|
217
|
-
reasons: CrossSurfacePairIncompatibilityReason[];
|
|
218
|
-
betterSingleCandidateId: string;
|
|
219
|
-
}
|
|
220
|
-
interface CrossSurfacePairwiseEntry {
|
|
221
|
-
componentIds: [string, string];
|
|
222
|
-
singleCandidateIds: [string, string];
|
|
223
|
-
compositionCandidateId: string;
|
|
224
|
-
benefitTaskIds: string[];
|
|
225
|
-
regressionTaskIds: string[];
|
|
226
|
-
synergyTaskIds: string[];
|
|
227
|
-
interferenceTaskIds: string[];
|
|
228
|
-
incrementalVsConstituents: [CrossSurfaceCandidateComparison, CrossSurfaceCandidateComparison];
|
|
229
|
-
relativeCostToBaseline: Record<string, CrossSurfaceRelativeCost>;
|
|
230
|
-
firing: CrossSurfacePairEvidence;
|
|
231
|
-
effect: CrossSurfacePairEvidence;
|
|
232
|
-
interaction: CrossSurfaceInteractionEffect;
|
|
233
|
-
compatibility: CrossSurfacePairCompatibility;
|
|
234
|
-
}
|
|
235
|
-
interface CrossSurfaceRankedSingle {
|
|
236
|
-
rank: number;
|
|
237
|
-
candidateId: string;
|
|
238
|
-
componentId: string;
|
|
239
|
-
}
|
|
240
|
-
interface CrossSurfaceBestSingleSelection {
|
|
241
|
-
candidateId: string;
|
|
242
|
-
componentId: string;
|
|
243
|
-
ranking: CrossSurfaceRankedSingle[];
|
|
244
|
-
}
|
|
245
|
-
interface CrossSurfaceNaiveStackSelection {
|
|
246
|
-
/** Every individually eligible single, stacked in canonical component order. */
|
|
247
|
-
candidateId: string;
|
|
248
|
-
componentIds: string[];
|
|
249
|
-
}
|
|
250
|
-
type CrossSurfaceAdditionRejectionReason = 'pair_incompatible' | 'full_bundle_not_evaluated' | 'bundle_incomplete' | 'baseline_regression' | 'no_incremental_resolution' | 'incremental_regression' | 'firing_below_minimum' | 'firing_unobserved' | 'effect_below_minimum' | 'effect_unobserved' | 'cost_limit_exceeded';
|
|
251
|
-
interface CrossSurfaceAdditionDecision {
|
|
252
|
-
additionCandidateId: string;
|
|
253
|
-
additionComponentId: string;
|
|
254
|
-
bundleCandidateId: string | null;
|
|
255
|
-
incrementalResolutionTaskIds: string[];
|
|
256
|
-
incrementalRegressionTaskIds: string[];
|
|
257
|
-
incrementalMedianCost: Record<string, number> | null;
|
|
258
|
-
eligible: boolean;
|
|
259
|
-
selected: boolean;
|
|
260
|
-
reasons: CrossSurfaceAdditionRejectionReason[];
|
|
261
|
-
}
|
|
262
|
-
interface CrossSurfaceCompositionStep {
|
|
263
|
-
fromCandidateId: string;
|
|
264
|
-
retainedComponentIds: string[];
|
|
265
|
-
considered: CrossSurfaceAdditionDecision[];
|
|
266
|
-
selectedCandidateId: string | null;
|
|
267
|
-
}
|
|
268
|
-
/** One deterministic growth path starting from a compatible two-surface seed. */
|
|
269
|
-
interface CrossSurfaceInteractionPath {
|
|
270
|
-
seedCandidateId: string;
|
|
271
|
-
terminalCandidateId: string;
|
|
272
|
-
terminalComponentIds: string[];
|
|
273
|
-
qualified: boolean;
|
|
274
|
-
steps: CrossSurfaceCompositionStep[];
|
|
275
|
-
}
|
|
276
|
-
interface CrossSurfaceInteractionAwareSelection {
|
|
277
|
-
/** Compatible pair that seeded the selected deterministic growth path. */
|
|
278
|
-
seedCandidateId: string;
|
|
279
|
-
/** Candidate reached by the winning path, even if the minimum size is not met. */
|
|
280
|
-
terminalCandidateId: string;
|
|
281
|
-
terminalComponentIds: string[];
|
|
282
|
-
/** null when no path produced a qualifying multi-component bundle. */
|
|
283
|
-
selectedCandidateId: string | null;
|
|
284
|
-
qualified: boolean;
|
|
285
|
-
/** Every compatible pair seed is retained so seed choice cannot hide an interaction. */
|
|
286
|
-
evaluatedPaths: CrossSurfaceInteractionPath[];
|
|
287
|
-
/** Convenience alias for the winning path's steps. */
|
|
288
|
-
steps: CrossSurfaceCompositionStep[];
|
|
289
|
-
}
|
|
290
|
-
interface CrossSurfaceSelections {
|
|
291
|
-
bestSingle: CrossSurfaceBestSingleSelection | null;
|
|
292
|
-
naiveStack: CrossSurfaceNaiveStackSelection | null;
|
|
293
|
-
interactionAware: CrossSurfaceInteractionAwareSelection | null;
|
|
294
|
-
}
|
|
295
|
-
interface CrossSurfaceInteractionReport<TRow extends CrossSurfaceTaskRow = CrossSurfaceTaskRow> {
|
|
296
|
-
taskIds: string[];
|
|
297
|
-
componentIds: string[];
|
|
298
|
-
candidateIds: string[];
|
|
299
|
-
costMetrics: string[];
|
|
300
|
-
/** Canonical candidate × task order; no input row is dropped. */
|
|
301
|
-
rows: TRow[];
|
|
302
|
-
missingAttempts: TRow[];
|
|
303
|
-
invalidAttempts: TRow[];
|
|
304
|
-
candidates: CrossSurfaceCandidateSummary[];
|
|
305
|
-
pairwise: CrossSurfacePairwiseEntry[];
|
|
306
|
-
selections: CrossSurfaceSelections;
|
|
307
|
-
}
|
|
308
|
-
//#endregion
|
|
309
|
-
//#region src/campaign/cross-surface-interaction.d.ts
|
|
310
|
-
/**
|
|
311
|
-
* Build the complete cross-surface evidence matrix and derive all three frozen
|
|
312
|
-
* candidates. The task/candidate/component orders are part of the input so
|
|
313
|
-
* neither insertion order nor an after-the-fact tie-break can change a result.
|
|
314
|
-
*/
|
|
315
|
-
declare function analyzeCrossSurfaceInteractions<TRow extends CrossSurfaceTaskRow>(input: AnalyzeCrossSurfaceInteractionsInput<TRow>): CrossSurfaceInteractionReport<TRow>;
|
|
316
|
-
//#endregion
|
|
317
|
-
//#region src/campaign/fixtures.d.ts
|
|
318
|
-
type EvalFixtureValidationMode = 'vitest' | 'none';
|
|
319
|
-
interface EvalFixtureFile {
|
|
320
|
-
path: string;
|
|
321
|
-
sha256: string;
|
|
322
|
-
bytes: number;
|
|
323
|
-
}
|
|
324
|
-
interface EvalFixture {
|
|
325
|
-
name: string;
|
|
326
|
-
path: string;
|
|
327
|
-
promptPath: string;
|
|
328
|
-
evalPath?: string;
|
|
329
|
-
packageJsonPath?: string;
|
|
330
|
-
prompt: string;
|
|
331
|
-
files: EvalFixtureFile[];
|
|
332
|
-
fingerprint: string;
|
|
333
|
-
}
|
|
334
|
-
interface EvalFixtureScenario extends Scenario {
|
|
335
|
-
kind: 'eval-fixture';
|
|
336
|
-
fixtureName: string;
|
|
337
|
-
fixturePath: string;
|
|
338
|
-
promptPath: string;
|
|
339
|
-
evalPath?: string;
|
|
340
|
-
packageJsonPath?: string;
|
|
341
|
-
prompt: string;
|
|
342
|
-
fingerprint: string;
|
|
343
|
-
}
|
|
344
|
-
interface EvalFixtureLoadOptions {
|
|
345
|
-
/** `vitest` requires EVAL.ts/EVAL.tsx and package.json type=module. `none` only requires PROMPT.md. */
|
|
346
|
-
validation?: EvalFixtureValidationMode;
|
|
347
|
-
/** Extra caller-owned knobs that affect fixture behavior, folded into the fingerprint. */
|
|
348
|
-
fingerprintConfig?: unknown;
|
|
349
|
-
}
|
|
350
|
-
interface LoadEvalFixtureScenariosOptions extends EvalFixtureLoadOptions {
|
|
351
|
-
names?: string[];
|
|
352
|
-
}
|
|
353
|
-
interface PlanEvalFixtureRunOptions<TArtifact = unknown> extends Pick<PlanCampaignRunOptions<EvalFixtureScenario, TArtifact>, 'dispatchRef' | 'judges' | 'seed' | 'reps' | 'resumable' | 'runDir'> {
|
|
354
|
-
evalsDir: string;
|
|
355
|
-
validation?: EvalFixtureValidationMode;
|
|
356
|
-
fingerprintConfig?: unknown;
|
|
357
|
-
names?: string[];
|
|
358
|
-
storage?: CampaignStorage;
|
|
359
|
-
}
|
|
360
|
-
type EvalFixtureRunPlan = CampaignRunPlan & {
|
|
361
|
-
fixtures: Array<Pick<EvalFixtureScenario, 'fixtureName' | 'fixturePath' | 'fingerprint'>>;
|
|
362
|
-
};
|
|
363
|
-
/** Walk `evalsDir` and return the relative name of every fixture directory (one containing an exact-case `PROMPT.md`). */
|
|
364
|
-
declare function discoverEvalFixtures(evalsDir: string): string[];
|
|
365
|
-
/**
|
|
366
|
-
* Load ONE fixture by name: reads `PROMPT.md` (plus `EVAL.ts`/`EVAL.tsx` and `package.json` under
|
|
367
|
-
* `vitest` validation) and content-fingerprints the full file set for cache identity.
|
|
368
|
-
*/
|
|
369
|
-
declare function loadEvalFixture(evalsDir: string, name: string, options?: EvalFixtureLoadOptions): EvalFixture;
|
|
370
|
-
/** Load fixtures (all discovered, or just `names`) as campaign `Scenario`s tagged `eval-fixture`. */
|
|
371
|
-
declare function loadEvalFixtureScenarios(evalsDir: string, options?: LoadEvalFixtureScenariosOptions): EvalFixtureScenario[];
|
|
372
|
-
/**
|
|
373
|
-
* Dry-run planner for a fixture campaign: loads the scenarios, delegates to `planCampaignRun`,
|
|
374
|
-
* and returns the plan plus each fixture's name/path/fingerprint.
|
|
375
|
-
*/
|
|
376
|
-
declare function planEvalFixtureRun<TArtifact = unknown>(options: PlanEvalFixtureRunOptions<TArtifact>): EvalFixtureRunPlan;
|
|
377
|
-
//#endregion
|
|
378
|
-
//#region src/campaign/gates/neutralization-gate.d.ts
|
|
379
|
-
interface NeutralizationGateOptions<TScenario extends Scenario = Scenario> {
|
|
380
|
-
scenarios: TScenario[];
|
|
381
|
-
/** Reject when the neutralized (content-blanked, footprint-matched) variant
|
|
382
|
-
* reproduces at least this fraction of the candidate's held-out lift. Default
|
|
383
|
-
* 0.5 — if blanking the content keeps half the lift, the content is decorative.
|
|
384
|
-
* Equality rejects: a neutralized lift == threshold·candidateLift is decorative. */
|
|
385
|
-
maxDecorativeFraction?: number;
|
|
386
|
-
}
|
|
387
|
-
/**
|
|
388
|
-
* Composable placebo gate: ships only when the candidate's held-out lift is NOT
|
|
389
|
-
* mostly reproduced by a footprint-matched neutralized variant.
|
|
390
|
-
*/
|
|
391
|
-
declare function neutralizationGate<TArtifact, TScenario extends Scenario>(options: NeutralizationGateOptions<TScenario>): Gate<TArtifact, TScenario>;
|
|
392
|
-
//#endregion
|
|
393
|
-
//#region src/campaign/grounded-reflection.d.ts
|
|
394
|
-
/**
|
|
395
|
-
* Evidence grounding for reflective optimizers (GEPA-style revise loops).
|
|
396
|
-
*
|
|
397
|
-
* Two failure modes recur when an LLM revises an artifact from raw rollout
|
|
398
|
-
* traces (first measured in agent-lab R358, where naive reflection REGRESSED
|
|
399
|
-
* the score 0.375 -> 0.125 before these helpers fixed it):
|
|
400
|
-
*
|
|
401
|
-
* 1. The environment often hides WHY a rollout failed - a tool call can
|
|
402
|
-
* succeed while an invisible downstream check fails - so the reviser
|
|
403
|
-
* cannot see the cause in the transcript. The only reliable signal is the
|
|
404
|
-
* field-level difference between what passing and failing rollouts did.
|
|
405
|
-
* `rolloutArgumentDiff` computes that difference deterministically so the
|
|
406
|
-
* reviser is handed the diff instead of being trusted to derive it.
|
|
407
|
-
*
|
|
408
|
-
* 2. Revisers invent plausible-but-wrong literal values ("use 'new'",
|
|
409
|
-
* "use 'sent'") that no passing rollout ever used, turning every rollout
|
|
410
|
-
* into a failure. `classifyUngroundedLiterals` mechanically detects them,
|
|
411
|
-
* separating HARMFUL literals (ones failing rollouts actually used -
|
|
412
|
-
* proven damage) from benign illustrations (e.g. a name example like
|
|
413
|
-
* 'Doe'), so callers can hard-reject the former and merely log the latter.
|
|
414
|
-
* Rejecting every ungrounded quoted word is too blunt: it killed a run
|
|
415
|
-
* over a surname illustration before the severity split existed.
|
|
416
|
-
*
|
|
417
|
-
* Pure data in, data out: no LLM calls, no filesystem, no domain knowledge.
|
|
418
|
-
*/
|
|
419
|
-
/** One tool/action call observed in a rollout: a name plus its arguments. */
|
|
420
|
-
interface RolloutCall {
|
|
421
|
-
readonly name: string;
|
|
422
|
-
readonly args: Readonly<Record<string, unknown>>;
|
|
423
|
-
}
|
|
424
|
-
/** A scored rollout: its calls plus the scalar outcome used to split pass/fail. */
|
|
425
|
-
interface ScoredRollout {
|
|
426
|
-
/** Caller-meaningful identifier (task id, cell id) used only for reporting. */
|
|
427
|
-
readonly id: string;
|
|
428
|
-
/** Scalar outcome in [0, 1]; `passThreshold` splits passing from failing. */
|
|
429
|
-
readonly score: number;
|
|
430
|
-
readonly calls: readonly RolloutCall[];
|
|
431
|
-
}
|
|
432
|
-
interface RolloutArgumentDiffOptions {
|
|
433
|
-
/** Rollouts with `score >= passThreshold` count as passing. Default 1. */
|
|
434
|
-
readonly passThreshold?: number;
|
|
435
|
-
/** Max distinct values listed per field per side in the rendered text. Default 4. */
|
|
436
|
-
readonly maxValuesPerField?: number;
|
|
437
|
-
}
|
|
438
|
-
interface RolloutArgumentDiff {
|
|
439
|
-
/** Human/LLM-readable per-field diff, one line per field. */
|
|
440
|
-
readonly text: string;
|
|
441
|
-
/** Lowercased stringified argument values seen in passing rollouts. */
|
|
442
|
-
readonly passingValues: ReadonlySet<string>;
|
|
443
|
-
/** Lowercased stringified argument values seen in failing rollouts. */
|
|
444
|
-
readonly failingValues: ReadonlySet<string>;
|
|
445
|
-
}
|
|
446
|
-
/**
|
|
447
|
-
* Deterministic per-field diff of call arguments between passing and failing
|
|
448
|
-
* rollouts. A field set by failing rollouts but left unset by passing ones is
|
|
449
|
-
* the classic poison-input signature; a field whose values differ across the
|
|
450
|
-
* split points at the correct value. Feed `text` to the reviser verbatim.
|
|
451
|
-
*/
|
|
452
|
-
declare function rolloutArgumentDiff(rollouts: readonly ScoredRollout[], opts?: RolloutArgumentDiffOptions): RolloutArgumentDiff;
|
|
453
|
-
interface UngroundedLiteralReport {
|
|
454
|
-
/** Quoted single-word literals in the text that no passing rollout used. */
|
|
455
|
-
readonly ungrounded: readonly string[];
|
|
456
|
-
/** The subset failing rollouts actually used - prescribing these is proven harmful. */
|
|
457
|
-
readonly harmful: readonly string[];
|
|
458
|
-
}
|
|
459
|
-
/**
|
|
460
|
-
* Scan revised artifact text for single-quoted single-word literals (the
|
|
461
|
-
* "use exactly 'new'" pattern) that appear in no passing rollout's argument
|
|
462
|
-
* values. Multi-word quotes pass (they are prose, not prescriptions).
|
|
463
|
-
* Callers should reject on `harmful` (with a bounded retry) and at most log
|
|
464
|
-
* `ungrounded` - see the module header for why the severities differ.
|
|
465
|
-
*/
|
|
466
|
-
declare function classifyUngroundedLiterals(text: string, diff: Pick<RolloutArgumentDiff, 'passingValues' | 'failingValues'>): UngroundedLiteralReport;
|
|
467
|
-
//#endregion
|
|
468
|
-
//#region src/campaign/labeled-store/fs-adapter.d.ts
|
|
469
|
-
interface FsLabeledScenarioStoreOptions {
|
|
470
|
-
/** Root directory for JSONL files. Created if missing. */
|
|
471
|
-
root: string;
|
|
472
|
-
/** Per-source rate limit. When set, writes exceeding the cap are rejected
|
|
473
|
-
* with a typed error. Default: no limit. */
|
|
474
|
-
maxWritesPerMinutePerBucket?: number;
|
|
475
|
-
/** Test seam — override `Date.now()` for deterministic tests. */
|
|
476
|
-
now?: () => number;
|
|
477
|
-
}
|
|
478
|
-
/** Typed rejection from a labeled-scenario store (bad provenance, rate limit, invalid sample args) — carries a stable string `code`. */
|
|
479
|
-
declare class LabeledScenarioStoreError extends Error {
|
|
480
|
-
readonly code: string;
|
|
481
|
-
constructor(code: string, message: string);
|
|
482
|
-
}
|
|
483
|
-
/**
|
|
484
|
-
* Filesystem `LabeledScenarioStore`: appends one JSONL file per source with provenance and
|
|
485
|
-
* rate-limit guards. For tests, local dev, and small workloads — high-throughput lands in Turso.
|
|
486
|
-
*/
|
|
487
|
-
declare class FsLabeledScenarioStore implements LabeledScenarioStore {
|
|
488
|
-
private readonly options;
|
|
489
|
-
private readonly now;
|
|
490
|
-
private readonly rateLimits;
|
|
491
|
-
constructor(options: FsLabeledScenarioStoreOptions);
|
|
492
|
-
observe(write: LabeledScenarioWrite): Promise<void>;
|
|
493
|
-
sample(args: LabeledScenarioSampleArgs): Promise<LabeledScenarioRecord[]>;
|
|
494
|
-
size(): Promise<{
|
|
495
|
-
train: number;
|
|
496
|
-
test: number;
|
|
497
|
-
bySource: Record<string, number>;
|
|
498
|
-
byTrust: Record<LabelTrust, number>;
|
|
499
|
-
}>;
|
|
500
|
-
private assertProvenance;
|
|
501
|
-
private assertRateLimit;
|
|
502
|
-
private toRecord;
|
|
503
|
-
private pathForSource;
|
|
504
|
-
}
|
|
505
|
-
//#endregion
|
|
506
|
-
//#region src/campaign/neutralize.d.ts
|
|
507
|
-
/**
|
|
508
|
-
* @module
|
|
509
|
-
* Footprint-matched neutralization — the placebo control for content-vs-footprint
|
|
510
|
-
* attribution in a promotion gate.
|
|
511
|
-
*
|
|
512
|
-
* A promoted surface can raise a held-out score two different ways:
|
|
513
|
-
* 1. its CONTENT is informative (the thing we want to promote), or
|
|
514
|
-
* 2. it merely added prompt/mount FOOTPRINT — more bytes, more lines, a longer
|
|
515
|
-
* more authoritative-looking prompt — that the model spends attention on
|
|
516
|
-
* regardless of what the bytes say.
|
|
517
|
-
*
|
|
518
|
-
* A held-out gate proves the candidate beat baseline; it cannot separate (1) from
|
|
519
|
-
* (2). `neutralizeText` produces a variant that keeps the input's layout and
|
|
520
|
-
* length while carrying ZERO information, so scoring it isolates the footprint
|
|
521
|
-
* contribution (2). Feed the neutralized variant's scores to `neutralizationGate`:
|
|
522
|
-
* any lift it still holds over baseline is decorative, and a candidate whose lift
|
|
523
|
-
* survives neutralization is rejected however large its raw lift.
|
|
524
|
-
*/
|
|
525
|
-
/**
|
|
526
|
-
* Blank every non-whitespace character to a 1-byte filler while preserving all
|
|
527
|
-
* whitespace. Line count, indentation, and word/line lengths are unchanged — so
|
|
528
|
-
* the neutralized variant has the same layout and (for ASCII) the same byte
|
|
529
|
-
* footprint as the input, but no readable content. Whitespace is preserved
|
|
530
|
-
* deliberately: collapsing it would change the token structure and stop the
|
|
531
|
-
* variant from being a true footprint match.
|
|
532
|
-
*/
|
|
533
|
-
declare function neutralizeText(content: string): string;
|
|
534
|
-
//#endregion
|
|
535
|
-
//#region src/campaign/presets/run-profile-matrix.d.ts
|
|
536
|
-
/** Thrown when the matrix is misconfigured (no profiles, missing resolved model evidence,
|
|
537
|
-
* etc.). Distinct from `BackendIntegrityError`,
|
|
538
|
-
* which signals a stub backend at run time. */
|
|
539
|
-
declare class ProfileMatrixError extends AgentEvalError {
|
|
540
|
-
constructor(message: string);
|
|
541
|
-
}
|
|
542
|
-
/** Dispatch for one cell: render `profile` against `scenario`, returning the
|
|
543
|
-
* artifact the judges score. Run LLM work through `ctx.cost.runPaidCall` —
|
|
544
|
-
* the integrity check depends on its receipt. */
|
|
545
|
-
type ProfileDispatchFn<TScenario extends Scenario, TArtifact> = (profile: AgentProfile$1, scenario: TScenario, ctx: DispatchContext) => Promise<TArtifact>;
|
|
546
|
-
interface RunProfileMatrixOptions<TScenario extends Scenario, TArtifact> {
|
|
547
|
-
/** Axis 3 — the agent-under-test configurations. Each is one column. */
|
|
548
|
-
profiles: AgentProfile$1[];
|
|
549
|
-
/** Axis 1 — the persona/scenario corpus, run against every profile. */
|
|
550
|
-
scenarios: TScenario[];
|
|
551
|
-
/** Renders one (profile, scenario) cell. */
|
|
552
|
-
dispatch: ProfileDispatchFn<TScenario, TArtifact>;
|
|
553
|
-
/** The scoring axis. */
|
|
554
|
-
judges?: JudgeConfig<TArtifact, TScenario>[];
|
|
555
|
-
/** Where each profile's campaign writes artifacts/traces. One subdir per
|
|
556
|
-
* profile. */
|
|
557
|
-
runDir: string;
|
|
558
|
-
/** Git SHA the harness ran from — stamped onto every RunRecord (mandatory
|
|
559
|
-
* for paper-grade records). */
|
|
560
|
-
commitSha: string;
|
|
561
|
-
/** Additional stable identity for dispatch behavior that can change without
|
|
562
|
-
* changing `commitSha`, such as a caller-owned executable or remote config. */
|
|
563
|
-
dispatchRef?: string;
|
|
564
|
-
/** Logical experiment id shared across the whole matrix so the promotion
|
|
565
|
-
* gate can pair profiles on matched scenarios. Default: a hash of the
|
|
566
|
-
* profile + scenario ids. */
|
|
567
|
-
experimentId?: string;
|
|
568
|
-
/** Which split these runs belong to. Default `'search'`. */
|
|
569
|
-
splitTag?: RunSplitTag;
|
|
570
|
-
/** Replicates per (profile, scenario) cell for CI bands. Default 1. */
|
|
571
|
-
reps?: number;
|
|
572
|
-
/** Campaign seed (per profile). Default 42. */
|
|
573
|
-
seed?: number;
|
|
574
|
-
/**
|
|
575
|
-
* Backend-integrity posture, enforced AFTER the matrix completes:
|
|
576
|
-
* - `'assert'` (default) — throw `BackendIntegrityError` if the run was a
|
|
577
|
-
* stub (and, with `allowMixed:false`, if it was mixed).
|
|
578
|
-
* - `'warn'` — log the verdict but never throw.
|
|
579
|
-
* - `'off'` — skip the guard entirely (only for offline/replay analysis).
|
|
580
|
-
*/
|
|
581
|
-
integrity?: 'assert' | 'warn' | 'off';
|
|
582
|
-
/** Forwarded to `assertRealBackend`. Default true (tolerate partial 429
|
|
583
|
-
* cascades); set false for strict CI gates. */
|
|
584
|
-
allowMixed?: boolean;
|
|
585
|
-
/** Max concurrent cells WITHIN each profile's campaign. Default 2. */
|
|
586
|
-
maxConcurrency?: number;
|
|
587
|
-
/** Max profile campaigns in flight. Default 1. Each profile keeps its own
|
|
588
|
-
* run directory and cost ceiling; raise this when those resources are independent. */
|
|
589
|
-
maxProfileConcurrency?: number;
|
|
590
|
-
/** Cumulative USD cap per profile campaign. */
|
|
591
|
-
costCeiling?: number;
|
|
592
|
-
/** Capture flywheel — forwarded to each campaign. */
|
|
593
|
-
labeledStore?: LabeledScenarioStore | 'off';
|
|
594
|
-
captureSource?: LabeledScenarioSource;
|
|
595
|
-
/** Storage backend. Default `fsCampaignStorage`. Pass
|
|
596
|
-
* `inMemoryCampaignStorage()` for edge/CF-Worker/test runs. */
|
|
597
|
-
storage?: CampaignStorage;
|
|
598
|
-
/** Test seam — override the wall clock. */
|
|
599
|
-
now?: () => Date;
|
|
600
|
-
/** Optional persona key per scenario — drives the `byPersona` pivot. When
|
|
601
|
-
* unset, `byPersona` is omitted. */
|
|
602
|
-
personaOf?: (scenario: TScenario) => string;
|
|
603
|
-
/** Validate every produced RunRecord with `validateRunRecord` (fail-loud).
|
|
604
|
-
* Default true — catches bad model snapshots and non-finite judge dims at
|
|
605
|
-
* the boundary instead of letting them poison downstream analysis. */
|
|
606
|
-
validate?: boolean;
|
|
607
|
-
/** Corpus-by-default: derive the trajectory text (`prompt` + `completion`)
|
|
608
|
-
* for each cell from its artifact + scenario. When set, every produced
|
|
609
|
-
* record carries `prompt`/`completion` (a `CorpusRecord`) so the run's
|
|
610
|
-
* graded trajectories can be appended to the durable RL corpus with no
|
|
611
|
-
* side-channel — `appendToCorpus(result.records, path)`. Fail-soft: a
|
|
612
|
-
* throwing or undefined-returning extractor just omits the text. */
|
|
613
|
-
corpusText?: (artifact: TArtifact, scenario: TScenario) => {
|
|
614
|
-
prompt: string;
|
|
615
|
-
completion: string;
|
|
616
|
-
} | undefined;
|
|
617
|
-
/**
|
|
618
|
-
* Optional explicit row selection. The matrix identity remains based on the
|
|
619
|
-
* complete profiles × scenarios × reps design; this invocation executes only
|
|
620
|
-
* rows accepted by the predicate.
|
|
621
|
-
*/
|
|
622
|
-
rowFilter?: (input: {
|
|
623
|
-
profile: AgentProfile$1;
|
|
624
|
-
scenario: TScenario;
|
|
625
|
-
rep: number;
|
|
626
|
-
}) => boolean;
|
|
627
|
-
/** Stable matrix identity supplied by a persisted profile-matrix plan. */
|
|
628
|
-
matrixId?: string;
|
|
629
|
-
/** Reuse cached failed cells. Normal matrix runs retry them by default. */
|
|
630
|
-
reuseFailedCells?: boolean;
|
|
631
|
-
}
|
|
632
|
-
interface ProfileSummary {
|
|
633
|
-
profileId: string;
|
|
634
|
-
profileHash: string;
|
|
635
|
-
model: string;
|
|
636
|
-
/** RunRecords produced for this profile (= scenarios × reps). */
|
|
637
|
-
records: number;
|
|
638
|
-
/** Mean across scored records, or null when the profile has no task labels. */
|
|
639
|
-
meanComposite: number | null;
|
|
640
|
-
/** Total cost, or null when any call's cost was not captured. */
|
|
641
|
-
totalCostUsd: number | null;
|
|
642
|
-
costProvenance: CostProvenance;
|
|
643
|
-
/** Per-profile integrity verdict — surfaces a single profile that ran stub
|
|
644
|
-
* even when the matrix as a whole looks real. */
|
|
645
|
-
integrity: BackendIntegrityReport;
|
|
646
|
-
}
|
|
647
|
-
interface ScenarioRollup {
|
|
648
|
-
meanComposite: number;
|
|
649
|
-
n: number;
|
|
650
|
-
}
|
|
651
|
-
interface RunProfileMatrixResult<TArtifact, TScenario extends Scenario> {
|
|
652
|
-
matrixId: string;
|
|
653
|
-
experimentId: string;
|
|
654
|
-
/** One RunRecord per (profile, scenario, rep) cell — the integrity-checked,
|
|
655
|
-
* paper-grade output. Feed straight into `analyzeRuns`, `HeldOutGate`,
|
|
656
|
-
* scorecards, the hosted wire format. */
|
|
657
|
-
records: RunRecord[];
|
|
658
|
-
byProfile: Record<string, ProfileSummary>;
|
|
659
|
-
byScenario: Record<string, ScenarioRollup>;
|
|
660
|
-
/** Present only when `personaOf` was supplied. */
|
|
661
|
-
byPersona?: Record<string, ScenarioRollup>;
|
|
662
|
-
/** Whole-matrix integrity report (the one `integrity:'assert'` enforces). */
|
|
663
|
-
integrity: BackendIntegrityReport;
|
|
664
|
-
/** The raw per-profile campaign results, keyed by profile id. */
|
|
665
|
-
campaigns: Record<string, CampaignResult<TArtifact, TScenario>>;
|
|
666
|
-
}
|
|
667
|
-
/**
|
|
668
|
-
* Profile × scenario matrix runner: fan N agent profiles across M scenarios, project each cell to a validated `RunRecord` with real token usage, and enforce the backend-integrity guard before returning.
|
|
669
|
-
*/
|
|
670
|
-
declare function runProfileMatrix<TScenario extends Scenario, TArtifact>(opts: RunProfileMatrixOptions<TScenario, TArtifact>): Promise<RunProfileMatrixResult<TArtifact, TScenario>>;
|
|
671
|
-
//#endregion
|
|
672
|
-
//#region src/campaign/presets/playback.d.ts
|
|
673
|
-
/** One step of a user story — what the user does. The driver interprets
|
|
674
|
-
* `payload` (a Playwright selector + action, or a sandbox chat turn). */
|
|
675
|
-
interface PlaybackStep {
|
|
676
|
-
/** Human-readable action, captured verbatim in the UX narrative. */
|
|
677
|
-
action: string;
|
|
678
|
-
/** Driver-specific payload (e.g. `{ selector, fill }` or `{ turn }`). */
|
|
679
|
-
payload?: Record<string, unknown>;
|
|
680
|
-
}
|
|
681
|
-
/**
|
|
682
|
-
* A user story = a runnable product journey plus the requirements that define
|
|
683
|
-
* "this story works". Each requirement is one Jira ticket line. Extends
|
|
684
|
-
* `Scenario` so a catalog drops straight into `runProfileMatrix({ scenarios })`.
|
|
685
|
-
*/
|
|
686
|
-
interface UserStory extends Scenario {
|
|
687
|
-
/** Human-readable story title (the ticket headline). */
|
|
688
|
-
title: string;
|
|
689
|
-
/** Ordered steps the driver executes. */
|
|
690
|
-
steps: PlaybackStep[];
|
|
691
|
-
/** What must hold in the produced state for the story to pass. */
|
|
692
|
-
requirements: CompletionRequirement[];
|
|
693
|
-
}
|
|
694
|
-
/** Dispatch context plus the profile under test (which cheap model, etc.). */
|
|
695
|
-
interface PlaybackContext extends DispatchContext {
|
|
696
|
-
profile: AgentProfile;
|
|
697
|
-
}
|
|
698
|
-
/**
|
|
699
|
-
* Drives the real product through a story and returns the runtime event stream
|
|
700
|
-
* `extractProducedState` consumes. Implemented by CONSUMERS —
|
|
701
|
-
* `SandboxPlaybackDriver` (real API / sandbox workspace) and
|
|
702
|
-
* `PlaywrightPlaybackDriver` (real UI) — because they depend on runtime /
|
|
703
|
-
* browser infra the substrate must not import. The driver MUST report LLM
|
|
704
|
-
* usage through `ctx.cost.runPaidCall` so the backend-integrity check sees real
|
|
705
|
-
* tokens (a run that never reports tokens reads as a stub).
|
|
706
|
-
*/
|
|
707
|
-
interface PlaybackDriver<TStory extends UserStory = UserStory> {
|
|
708
|
-
run(story: TStory, ctx: PlaybackContext): Promise<readonly RuntimeEventLike[]>;
|
|
709
|
-
}
|
|
710
|
-
/**
|
|
711
|
-
* Adapt a `PlaybackDriver` into a `runProfileMatrix` dispatch. The artifact the
|
|
712
|
-
* matrix scores is the `ProducedState` extracted from the driver's event
|
|
713
|
-
* stream — grade it with `scoreUserStory` (or a judge wrapping it).
|
|
714
|
-
*/
|
|
715
|
-
declare function makePlaybackDispatch<TStory extends UserStory>(driver: PlaybackDriver<TStory>): ProfileDispatchFn<TStory, ProducedState>;
|
|
716
|
-
/** A scored user story — the completion verdict plus its human title. */
|
|
717
|
-
interface UserStoryVerdict extends CompletionVerdict {
|
|
718
|
-
title: string;
|
|
719
|
-
}
|
|
720
|
-
/**
|
|
721
|
-
* Score one story's produced state against its requirements. Thin wrapper over
|
|
722
|
-
* `verifyCompletion` that builds the gold from the story and returns a
|
|
723
|
-
* per-requirement PASS/FAIL verdict. `checkCorrectness` is injected — a
|
|
724
|
-
* deterministic stub in tests, `createLlmCorrectnessChecker` in production.
|
|
725
|
-
*/
|
|
726
|
-
declare function scoreUserStory(story: UserStory, state: ProducedState, checkCorrectness: CorrectnessChecker): Promise<UserStoryVerdict>;
|
|
727
|
-
/** One row of the launch scoreboard — story × requirement → PASS/FAIL. */
|
|
728
|
-
interface ScoreboardRow {
|
|
729
|
-
storyId: string;
|
|
730
|
-
storyTitle: string;
|
|
731
|
-
reqId: string;
|
|
732
|
-
reqTitle: string;
|
|
733
|
-
status: 'PASS' | 'FAIL';
|
|
734
|
-
evidence: string[];
|
|
735
|
-
}
|
|
736
|
-
/**
|
|
737
|
-
* Flatten story verdicts into the per-requirement scoreboard — the literal
|
|
738
|
-
* Jira tick-off: one row per (story, requirement) with PASS/FAIL and the
|
|
739
|
-
* evidence behind the verdict.
|
|
740
|
-
*/
|
|
741
|
-
declare function userStoryScoreboard(verdicts: readonly UserStoryVerdict[]): ScoreboardRow[];
|
|
742
|
-
/** Launch-readiness headline counts rolled up from the per-requirement rows. */
|
|
743
|
-
interface ScoreboardSummary {
|
|
744
|
-
/** Distinct user stories on the board. */
|
|
745
|
-
stories: number;
|
|
746
|
-
/** Stories whose every requirement passed. */
|
|
747
|
-
storiesFullyComplete: number;
|
|
748
|
-
/** Total (story, requirement) rows. */
|
|
749
|
-
requirements: number;
|
|
750
|
-
/** Rows with status PASS. */
|
|
751
|
-
passed: number;
|
|
752
|
-
/** Rows with status FAIL. */
|
|
753
|
-
failed: number;
|
|
754
|
-
/** passed / requirements; 0 when there are no rows. */
|
|
755
|
-
passRate: number;
|
|
756
|
-
}
|
|
757
|
-
/** Roll the per-requirement rows up into the launch headline counts. */
|
|
758
|
-
declare function scoreboardSummary(rows: readonly ScoreboardRow[]): ScoreboardSummary;
|
|
759
|
-
interface ScoreboardRenderOptions {
|
|
760
|
-
/** Document H1. Defaults to a generic playback title. */
|
|
761
|
-
title?: string;
|
|
762
|
-
/** Key/value run metadata rendered under the headline (runId, backend, model, date). */
|
|
763
|
-
meta?: Record<string, string>;
|
|
764
|
-
/** Max chars of joined evidence shown per row. Default 160. */
|
|
765
|
-
maxEvidenceChars?: number;
|
|
766
|
-
}
|
|
767
|
-
/**
|
|
768
|
-
* Render the scoreboard as a launch-readiness Markdown document — the literal
|
|
769
|
-
* "tick off every user story" artifact: a headline roll-up, the open tickets
|
|
770
|
-
* (FAIL rows) up top as the launch blockers, then a per-story table of
|
|
771
|
-
* requirement → PASS/FAIL with the evidence behind each verdict. Pure: same
|
|
772
|
-
* rows in, same bytes out (no clock/random), so it is safe to snapshot.
|
|
773
|
-
*/
|
|
774
|
-
declare function renderScoreboardMarkdown(rows: readonly ScoreboardRow[], opts?: ScoreboardRenderOptions): string;
|
|
775
|
-
//#endregion
|
|
776
|
-
//#region src/campaign/presets/segmented-profile-matrix.d.ts
|
|
777
|
-
interface ProfileMatrixRow {
|
|
778
|
-
rowId: string;
|
|
779
|
-
ordinal: number;
|
|
780
|
-
profileId: string;
|
|
781
|
-
scenarioId: string;
|
|
782
|
-
rep: number;
|
|
783
|
-
}
|
|
784
|
-
interface ProfileMatrixPlan<TScenario extends Scenario, TArtifact> {
|
|
785
|
-
readonly schemaVersion: 1;
|
|
786
|
-
readonly matrixId: string;
|
|
787
|
-
readonly experimentId: string;
|
|
788
|
-
readonly planDigest: `sha256:${string}`;
|
|
789
|
-
readonly profiles: readonly AgentProfile$1[];
|
|
790
|
-
readonly scenarios: readonly TScenario[];
|
|
791
|
-
readonly judges: readonly JudgeConfig<TArtifact, TScenario>[];
|
|
792
|
-
readonly reps: number;
|
|
793
|
-
readonly seed: number;
|
|
794
|
-
readonly splitTag: NonNullable<RunProfileMatrixOptions<TScenario, TArtifact>['splitTag']>;
|
|
795
|
-
readonly commitSha: string;
|
|
796
|
-
/** Stable dispatch implementation/configuration identity used by cell caches. */
|
|
797
|
-
readonly dispatchRef: string;
|
|
798
|
-
readonly integrity: NonNullable<RunProfileMatrixOptions<TScenario, TArtifact>['integrity']>;
|
|
799
|
-
readonly allowMixed: boolean;
|
|
800
|
-
readonly validate: boolean;
|
|
801
|
-
readonly personaOf?: (scenario: TScenario) => string;
|
|
802
|
-
readonly corpusText?: (artifact: TArtifact, scenario: TScenario) => {
|
|
803
|
-
prompt: string;
|
|
804
|
-
completion: string;
|
|
805
|
-
} | undefined;
|
|
806
|
-
readonly rows: readonly ProfileMatrixRow[];
|
|
807
|
-
}
|
|
808
|
-
interface CreateProfileMatrixPlanOptions<TScenario extends Scenario, TArtifact> extends Pick<RunProfileMatrixOptions<TScenario, TArtifact>, 'profiles' | 'scenarios' | 'judges' | 'commitSha' | 'experimentId' | 'splitTag' | 'reps' | 'seed' | 'integrity' | 'allowMixed' | 'validate' | 'personaOf' | 'corpusText'> {
|
|
809
|
-
/** Stable dispatch implementation/configuration identity used by cell caches. */
|
|
810
|
-
dispatchRef: string;
|
|
811
|
-
}
|
|
812
|
-
interface ProfileMatrixCoverage {
|
|
813
|
-
expected: number;
|
|
814
|
-
present: number;
|
|
815
|
-
missing: string[];
|
|
816
|
-
failed: string[];
|
|
817
|
-
zeroScore: string[];
|
|
818
|
-
}
|
|
819
|
-
interface RunProfileMatrixSegmentOptions<TScenario extends Scenario, TArtifact> {
|
|
820
|
-
plan: ProfileMatrixPlan<TScenario, TArtifact>;
|
|
821
|
-
/** Stable external grant or attempt identity. Reuse it to resume. */
|
|
822
|
-
segmentId: string;
|
|
823
|
-
/** Explicit row ids from `plan.rows`; duplicate or unknown rows fail. */
|
|
824
|
-
rows: readonly (string | ProfileMatrixRow)[];
|
|
825
|
-
dispatch: ProfileDispatchFn<TScenario, TArtifact>;
|
|
826
|
-
runDir: string;
|
|
827
|
-
storage?: CampaignStorage;
|
|
828
|
-
maxConcurrency?: number;
|
|
829
|
-
maxProfileConcurrency?: number;
|
|
830
|
-
costCeiling?: number;
|
|
831
|
-
labeledStore?: RunProfileMatrixOptions<TScenario, TArtifact>['labeledStore'];
|
|
832
|
-
captureSource?: RunProfileMatrixOptions<TScenario, TArtifact>['captureSource'];
|
|
833
|
-
now?: () => Date;
|
|
834
|
-
}
|
|
835
|
-
interface ProfileMatrixSegmentResult<TArtifact, TScenario extends Scenario> {
|
|
836
|
-
segmentId: string;
|
|
837
|
-
rowIds: string[];
|
|
838
|
-
matrix: RunProfileMatrixResult<TArtifact, TScenario>;
|
|
839
|
-
coverage: ProfileMatrixCoverage;
|
|
840
|
-
}
|
|
841
|
-
interface FinalizeProfileMatrixOptions<TScenario extends Scenario, TArtifact> {
|
|
842
|
-
plan: ProfileMatrixPlan<TScenario, TArtifact>;
|
|
843
|
-
runDir: string;
|
|
844
|
-
storage?: CampaignStorage;
|
|
845
|
-
maxConcurrency?: number;
|
|
846
|
-
maxProfileConcurrency?: number;
|
|
847
|
-
now?: () => Date;
|
|
848
|
-
}
|
|
849
|
-
interface FinalizedProfileMatrixResult<TArtifact, TScenario extends Scenario> extends RunProfileMatrixResult<TArtifact, TScenario> {
|
|
850
|
-
coverage: ProfileMatrixCoverage;
|
|
851
|
-
}
|
|
852
|
-
declare function createProfileMatrixPlan<TScenario extends Scenario, TArtifact>(opts: CreateProfileMatrixPlanOptions<TScenario, TArtifact>): ProfileMatrixPlan<TScenario, TArtifact>;
|
|
853
|
-
declare function runProfileMatrixSegment<TScenario extends Scenario, TArtifact>(opts: RunProfileMatrixSegmentOptions<TScenario, TArtifact>): Promise<ProfileMatrixSegmentResult<TArtifact, TScenario>>;
|
|
854
|
-
declare function finalizeProfileMatrix<TScenario extends Scenario, TArtifact>(opts: FinalizeProfileMatrixOptions<TScenario, TArtifact>): Promise<FinalizedProfileMatrixResult<TArtifact, TScenario>>;
|
|
855
|
-
//#endregion
|
|
856
|
-
//#region src/campaign/run-dir.d.ts
|
|
857
|
-
/** The shared, out-of-repo root for campaign/benchmark run bundles. Keeping run
|
|
858
|
-
* outputs here means they never land in a repo working tree (no per-repo
|
|
859
|
-
* gitignore, no clutter, no accidental commits). Layout:
|
|
860
|
-
* ~/.tangle/traces/<repo>/runs/<runName>/
|
|
861
|
-
* where <repo> disambiguates runs across repos in one place. */
|
|
862
|
-
declare function tangleTracesRoot(): string;
|
|
863
|
-
/** Resolve a campaign `runDir`. An absolute path is honored as-is (the caller
|
|
864
|
-
* chose an explicit location). A bare name is placed under the shared home root
|
|
865
|
-
* so bundles never pollute a repo working tree — the default the harness should
|
|
866
|
-
* compute so callers pass a *name*, not a path. */
|
|
867
|
-
declare function resolveRunDir(runDir: string, repo?: string): string;
|
|
868
|
-
//#endregion
|
|
869
|
-
//#region src/campaign/scenario-selection.d.ts
|
|
870
|
-
/**
|
|
871
|
-
* Discriminative scenario selection (research claim E2).
|
|
872
|
-
*
|
|
873
|
-
* The OR benchmark is SATURATING: run 7 measured ~75% tied holdout cells — most
|
|
874
|
-
* problems are solved optimally by the baseline AND every candidate, so those
|
|
875
|
-
* paired cells carry zero signal. A random/balanced holdout split spends its
|
|
876
|
-
* budget on scenarios that cannot separate candidates.
|
|
877
|
-
*
|
|
878
|
-
* This picks the holdout by DISCRIMINATION power instead: a scenario every
|
|
879
|
-
* candidate scores identically (variance ~0) carries no signal; one where the
|
|
880
|
-
* scores spread carries the most. We drop fully saturated ties so each paired
|
|
881
|
-
* holdout cell is spent on a scenario that can actually move a verdict.
|
|
882
|
-
*/
|
|
883
|
-
/** Per-scenario observation: the composite scores each candidate earned on it. */
|
|
884
|
-
interface ScenarioSignal {
|
|
885
|
-
scenarioId: string;
|
|
886
|
-
/** Per-candidate composite scores observed for this scenario (>=1 values). */
|
|
887
|
-
scores: number[];
|
|
888
|
-
}
|
|
889
|
-
interface DiscriminationScore {
|
|
890
|
-
scenarioId: string;
|
|
891
|
-
/** Higher = separates candidates more (spread of their scores). */
|
|
892
|
-
discrimination: number;
|
|
893
|
-
/** Higher = easier / more-saturated (mean of candidate scores). */
|
|
894
|
-
meanScore: number;
|
|
895
|
-
variance: number;
|
|
896
|
-
/** variance ~0 AND meanScore at/above the ceiling ⇒ a saturated tie, no signal. */
|
|
897
|
-
tied: boolean;
|
|
898
|
-
}
|
|
899
|
-
/**
|
|
900
|
-
* Rank scenarios by how well they DISCRIMINATE candidates.
|
|
901
|
-
*
|
|
902
|
-
* `discrimination = variance` (spread of the candidate scores) — kept simple on
|
|
903
|
-
* purpose; the headroom term (`saturationCeiling - meanScore`) only breaks ties
|
|
904
|
-
* so that, among equally spread scenarios, the one with more room to improve
|
|
905
|
-
* ranks first. Returned sorted by the deterministic order above.
|
|
906
|
-
*/
|
|
907
|
-
declare function scoreDiscrimination(signals: ScenarioSignal[], opts?: {
|
|
908
|
-
saturationCeiling?: number;
|
|
909
|
-
}): DiscriminationScore[];
|
|
910
|
-
/**
|
|
911
|
-
* Select the top-`k` most discriminative scenario ids for a holdout, EXCLUDING
|
|
912
|
-
* fully saturated ties when enough non-tied scenarios exist (a tie in the
|
|
913
|
-
* holdout wastes a paired cell).
|
|
914
|
-
*
|
|
915
|
-
* Prefers non-tied scenarios; if fewer than `k` non-tied exist, fills with the
|
|
916
|
-
* least-saturated tied ones (tied scenarios are already ordered least-saturated
|
|
917
|
-
* first by `meanScore` asc). Deterministic. Throws if `k < 1`. If
|
|
918
|
-
* `signals.length <= k`, returns all ids in discrimination order.
|
|
919
|
-
*/
|
|
920
|
-
declare function selectDiscriminative(signals: ScenarioSignal[], k: number, opts?: {
|
|
921
|
-
saturationCeiling?: number;
|
|
922
|
-
}): string[];
|
|
923
|
-
//#endregion
|
|
924
|
-
//#region src/campaign/score-utils.d.ts
|
|
925
|
-
/** Mean composite across cells with complete task-quality evidence.
|
|
926
|
-
* Partial judge results remain on their cells but never enter this value.
|
|
927
|
-
* A campaign with no complete score has no numeric mean and fails loudly. */
|
|
928
|
-
declare function campaignMeanComposite<TArtifact, TScenario extends Scenario>(campaign: CampaignResult<TArtifact, TScenario>): number;
|
|
929
|
-
/** Compare fixed-length lexicographic rank keys where each element is higher-is-better.
|
|
930
|
-
* Returns a positive number when `a` ranks above `b`, negative when below, and
|
|
931
|
-
* zero when equal. */
|
|
932
|
-
declare function compareRankKeys(a: readonly number[], b: readonly number[]): number;
|
|
933
|
-
interface CampaignBreakdown {
|
|
934
|
-
/** Mean score per judge dimension across all cells. */
|
|
935
|
-
dimensions: Record<string, number>;
|
|
936
|
-
/** Per-scenario composite (mean over reps + judges) + the judge's free-form
|
|
937
|
-
* `notes` for that scenario (the "why" a reflective proposer grounds on) +
|
|
938
|
-
* an optional `emitted` excerpt of the candidate's raw output (the "what it
|
|
939
|
-
* actually did" a reflective proposer grounds on). */
|
|
940
|
-
scenarios: Array<{
|
|
941
|
-
scenarioId: string;
|
|
942
|
-
composite: number;
|
|
943
|
-
notes?: string;
|
|
944
|
-
emitted?: string;
|
|
945
|
-
}>;
|
|
946
|
-
}
|
|
947
|
-
/** Per-candidate evidence a reflective/patch proposer grounds its next proposal
|
|
948
|
-
* on: mean score per judge dimension + per-scenario composite. */
|
|
949
|
-
declare function campaignBreakdown<TArtifact, TScenario extends Scenario>(campaign: CampaignResult<TArtifact, TScenario>): CampaignBreakdown;
|
|
950
|
-
//#endregion
|
|
951
|
-
//#region src/campaign/single-run-lock.d.ts
|
|
952
|
-
/**
|
|
953
|
-
* Single-run lock for evaluations that share one mutable environment.
|
|
954
|
-
*
|
|
955
|
-
* Two concurrent runs against a shared stateful gym silently corrupt each
|
|
956
|
-
* other: each resets/mutates environment state mid-cell of the other, and
|
|
957
|
-
* every score from both becomes garbage that LOOKS like worker variance
|
|
958
|
-
* (agent-lab R357 burned hours on flip-flopping scores before tracing them
|
|
959
|
-
* to exactly this). The fix is a pid lockfile: refuse to start while a live
|
|
960
|
-
* holder exists, reclaim stale locks whose pid is gone, release only if the
|
|
961
|
-
* lock is still ours.
|
|
962
|
-
*
|
|
963
|
-
* `alsoCheck` exists because independent runners can guard the same shared
|
|
964
|
-
* resource with differently named lockfiles; a runner must respect all of
|
|
965
|
-
* them even though it writes only its own.
|
|
966
|
-
*/
|
|
967
|
-
interface SingleRunLockOptions {
|
|
968
|
-
/** Lockfile this runner writes (and checks). */
|
|
969
|
-
readonly lockPath: string;
|
|
970
|
-
/** Other runners' lockfiles guarding the same resource; checked, never written. */
|
|
971
|
-
readonly alsoCheck?: readonly string[];
|
|
972
|
-
/** Install a process 'exit' hook that releases the lock. Default true. */
|
|
973
|
-
readonly releaseOnExit?: boolean;
|
|
974
|
-
/** Owner pid recorded in the lockfile metadata. Default process.pid. */
|
|
975
|
-
readonly pid?: number;
|
|
976
|
-
}
|
|
977
|
-
interface SingleRunLock {
|
|
978
|
-
/** Remove the lockfile if this process still owns it. Idempotent. */
|
|
979
|
-
release(): void;
|
|
980
|
-
}
|
|
981
|
-
/**
|
|
982
|
-
* Acquire the lock or throw naming the live holder. A stale lock (holder pid
|
|
983
|
-
* no longer running) is reclaimed by one contender. An interrupted reclaim
|
|
984
|
-
* leaves a marker that fails closed instead of admitting overlapping runs.
|
|
985
|
-
*/
|
|
986
|
-
declare function acquireSingleRunLock(opts: SingleRunLockOptions): SingleRunLock;
|
|
987
|
-
//#endregion
|
|
988
|
-
//#region src/campaign/surface-identity.d.ts
|
|
989
|
-
/** Validate the immutable identity shape; the owning executor verifies the Git objects and patch. */
|
|
990
|
-
declare function assertCodeSurfaceIdentity(surface: unknown): asserts surface is CodeSurface;
|
|
991
|
-
/**
|
|
992
|
-
* Deterministic identity material for a component surface.
|
|
993
|
-
*
|
|
994
|
-
* `canonicalString` orders keys by UTF-16 code unit (RFC 8785), which is a
|
|
995
|
-
* property of the value alone. The previous material ordered them with
|
|
996
|
-
* `localeCompare`, which reads the host's collation — so the same surface
|
|
997
|
-
* could produce two different identities on two machines, and the stored
|
|
998
|
-
* identity would stop matching a recomputation of the identical surface.
|
|
999
|
-
*/
|
|
1000
|
-
declare function componentSurfaceIdentityMaterial(surface: ComponentSurface): string;
|
|
1001
|
-
/** Canonical, location-independent identity of a finalized code candidate.
|
|
1002
|
-
* Commit metadata is excluded: two commits with the same base, final tree,
|
|
1003
|
-
* and patch bytes are the same executable candidate. */
|
|
1004
|
-
declare function codeSurfaceIdentityMaterial(surface: CodeSurface): string;
|
|
1005
|
-
/** Full SHA-256 content identity for a prompt or finalized code surface. */
|
|
1006
|
-
declare function surfaceContentHash(surface: MutableSurface): `sha256:${string}`;
|
|
1007
|
-
/** Short loop key derived from the same content identity as provenance. */
|
|
1008
|
-
declare function surfaceHash(surface: MutableSurface): string;
|
|
1009
|
-
/** Canonical customer-visible description of the exact before/after surfaces. */
|
|
1010
|
-
declare function renderSurfaceDiff(winnerSurface: MutableSurface, baselineSurface: MutableSurface): string;
|
|
1011
|
-
//#endregion
|
|
1012
|
-
//#region src/campaign/upstream-evaluators.d.ts
|
|
1013
|
-
interface PhoenixEvaluationResultLike {
|
|
1014
|
-
score?: number;
|
|
1015
|
-
label?: string;
|
|
1016
|
-
explanation?: string;
|
|
1017
|
-
}
|
|
1018
|
-
interface PhoenixEvaluatorLike<TRecord extends Record<string, unknown>> {
|
|
1019
|
-
name: string;
|
|
1020
|
-
kind: 'LLM' | 'CODE';
|
|
1021
|
-
optimizationDirection?: 'MAXIMIZE' | 'MINIMIZE' | 'NEUTRAL';
|
|
1022
|
-
evaluate(record: TRecord, context: UpstreamEvaluationContext): Promise<PhoenixEvaluationResultLike>;
|
|
1023
|
-
}
|
|
1024
|
-
interface AutoevalsScoreLike {
|
|
1025
|
-
name: string;
|
|
1026
|
-
score: number | null;
|
|
1027
|
-
metadata?: Record<string, unknown>;
|
|
1028
|
-
}
|
|
1029
|
-
type AutoevalsScorerLike<TInput extends Record<string, unknown>> = (input: TInput, context: UpstreamEvaluationContext) => AutoevalsScoreLike | Promise<AutoevalsScoreLike>;
|
|
1030
|
-
interface UpstreamEvaluationContext {
|
|
1031
|
-
readonly signal: AbortSignal;
|
|
1032
|
-
readonly callId?: string;
|
|
1033
|
-
}
|
|
1034
|
-
type PaidEvaluationOptions<TResult> = Pick<RunPaidCallInput<TResult>, 'maximumCharge' | 'receipt' | 'receiptFromError'> & {
|
|
1035
|
-
model: string;
|
|
1036
|
-
};
|
|
1037
|
-
interface UpstreamJudgeOptions<TScenario extends Scenario> {
|
|
1038
|
-
name?: string;
|
|
1039
|
-
dimension?: string;
|
|
1040
|
-
judgeVersion?: string;
|
|
1041
|
-
appliesTo?: (scenario: TScenario) => boolean;
|
|
1042
|
-
/** Convert the upstream score when its native scale is not higher-is-better. */
|
|
1043
|
-
toComposite?: (score: number) => number;
|
|
1044
|
-
}
|
|
1045
|
-
declare function phoenixEvaluatorJudge<TRecord extends Record<string, unknown>, TArtifact, TScenario extends Scenario = Scenario>(evaluator: PhoenixEvaluatorLike<TRecord>, options: UpstreamJudgeOptions<TScenario> & {
|
|
1046
|
-
mapInput(input: {
|
|
1047
|
-
artifact: TArtifact;
|
|
1048
|
-
scenario: TScenario;
|
|
1049
|
-
}): TRecord;
|
|
1050
|
-
paidCall?: PaidEvaluationOptions<PhoenixEvaluationResultLike>;
|
|
1051
|
-
}): JudgeConfig<TArtifact, TScenario>;
|
|
1052
|
-
declare function autoevalsScorerJudge<TInput extends Record<string, unknown>, TArtifact, TScenario extends Scenario = Scenario>(scorer: AutoevalsScorerLike<TInput>, options: UpstreamJudgeOptions<TScenario> & {
|
|
1053
|
-
name: string;
|
|
1054
|
-
mapInput(input: {
|
|
1055
|
-
artifact: TArtifact;
|
|
1056
|
-
scenario: TScenario;
|
|
1057
|
-
}): TInput;
|
|
1058
|
-
} & ({
|
|
1059
|
-
kind: 'CODE';
|
|
1060
|
-
paidCall?: never;
|
|
1061
|
-
} | {
|
|
1062
|
-
kind: 'LLM';
|
|
1063
|
-
paidCall: PaidEvaluationOptions<AutoevalsScoreLike>;
|
|
1064
|
-
})): JudgeConfig<TArtifact, TScenario>;
|
|
1065
|
-
//#endregion
|
|
1066
|
-
//#region src/campaign/worktree/index.d.ts
|
|
1067
|
-
type GitOutput = string | Uint8Array;
|
|
1068
|
-
type GitEnvironment = Readonly<Record<string, string>>;
|
|
1069
|
-
type GitRunner = (args: string[], cwd: string, env?: GitEnvironment) => GitOutput;
|
|
1070
|
-
interface Worktree {
|
|
1071
|
-
/** Absolute path to the checked-out worktree directory. */
|
|
1072
|
-
readonly path: string;
|
|
1073
|
-
/** The branch the worktree is on (becomes the PR branch on promotion). */
|
|
1074
|
-
readonly branch: string;
|
|
1075
|
-
/** The ref the worktree was forked from. */
|
|
1076
|
-
readonly baseRef: string;
|
|
1077
|
-
/** Exact commit `baseRef` resolved to before the worktree was created. */
|
|
1078
|
-
readonly baseCommit: string;
|
|
1079
|
-
/** Exact tree object for `baseCommit`. */
|
|
1080
|
-
readonly baseTree: string;
|
|
1081
|
-
}
|
|
1082
|
-
interface WorktreeAdapter {
|
|
1083
|
-
/** Create an isolated worktree on a fresh branch off `baseRef`. */
|
|
1084
|
-
create(opts: {
|
|
1085
|
-
baseRef: string;
|
|
1086
|
-
label: string;
|
|
1087
|
-
}): Promise<Worktree>;
|
|
1088
|
-
/** Commit pending changes, freeze the exact Git objects + binary patch, and
|
|
1089
|
-
* verify the worktree still matches that identity. */
|
|
1090
|
-
finalize(worktree: Worktree, summary: string): Promise<CodeSurface>;
|
|
1091
|
-
/** Idempotently remove the worktree and branch. Safe to retry after partial cleanup. */
|
|
1092
|
-
discard(worktree: Worktree): Promise<void>;
|
|
1093
|
-
}
|
|
1094
|
-
/** Typed failure from a `WorktreeAdapter` operation (create/finalize/discard) — wraps the underlying git error as `cause`. */
|
|
1095
|
-
declare class WorktreeAdapterError extends Error {
|
|
1096
|
-
readonly cause?: unknown;
|
|
1097
|
-
constructor(message: string, cause?: unknown);
|
|
1098
|
-
}
|
|
1099
|
-
interface GitWorktreeAdapterOptions {
|
|
1100
|
-
/** Repo root the worktrees fork from. */
|
|
1101
|
-
repoRoot: string;
|
|
1102
|
-
/** Directory worktrees are created under. Default: `<repoRoot>/.worktrees`. */
|
|
1103
|
-
worktreeDir?: string;
|
|
1104
|
-
/** Branch-name prefix. Default: `improve`. */
|
|
1105
|
-
branchPrefix?: string;
|
|
1106
|
-
/** Test seam — defaults to a real `git` runner. The return value must contain
|
|
1107
|
-
* stdout verbatim, and runners that execute Git must forward the optional
|
|
1108
|
-
* environment overrides used to isolate patch generation. */
|
|
1109
|
-
git?: GitRunner;
|
|
1110
|
-
}
|
|
1111
|
-
interface CodeSurfaceVerification {
|
|
1112
|
-
/** Verified worktree path. */
|
|
1113
|
-
path: string;
|
|
1114
|
-
/** Git's canonical root for the verified checkout. */
|
|
1115
|
-
repoRoot: string;
|
|
1116
|
-
/** Recomputed full content identity. */
|
|
1117
|
-
contentHash: `sha256:${string}`;
|
|
1118
|
-
/** Exact verified binary-patch bytes. Candidate-bundle builders encode this
|
|
1119
|
-
* directly instead of reproducing Git diff options. */
|
|
1120
|
-
patchBytes: Uint8Array;
|
|
1121
|
-
}
|
|
1122
|
-
/**
|
|
1123
|
-
* Git-backed `WorktreeAdapter`: creates isolated worktrees on fresh branches, commits agent changes, and discards losers.
|
|
1124
|
-
*/
|
|
1125
|
-
declare function gitWorktreeAdapter(opts: GitWorktreeAdapterOptions): WorktreeAdapter;
|
|
1126
|
-
/** Verify a finalized code surface against its current checkout. This rejects
|
|
1127
|
-
* dirty/ignored files, moved refs, missing Git objects, raw byte/mode
|
|
1128
|
-
* mismatches, external symlinks, and submodules. */
|
|
1129
|
-
declare function verifyCodeSurface(surface: CodeSurface, worktreeDir?: string): CodeSurfaceVerification;
|
|
1130
|
-
/** Resolve a code candidate for evaluation only after verifying its immutable
|
|
1131
|
-
* identity against the checkout at `worktreeRef`. */
|
|
1132
|
-
declare function resolveWorktreePath(surface: CodeSurface, worktreeDir?: string): string;
|
|
1133
|
-
//#endregion
|
|
1134
|
-
export { makePlaybackDispatch as $, CrossSurfaceEvidenceBreakdown as $t, scoreDiscrimination as A, LoadEvalFixtureScenariosOptions as At, ProfileMatrixSegmentResult as B, CrossSurfaceAttemptCompleteness as Bt, acquireSingleRunLock as C, SearchLedgerIntegrityError as Cn, neutralizationGate as Ct, compareRankKeys as D, EvalFixtureRunPlan as Dt, campaignMeanComposite as E, EvalFixtureLoadOptions as Et, FinalizeProfileMatrixOptions as F, planEvalFixtureRun as Ft, PlaybackContext as G, CrossSurfaceCandidateEvidence as Gt, createProfileMatrixPlan as H, CrossSurfaceBootstrapPolicy as Ht, FinalizedProfileMatrixResult as I, analyzeCrossSurfaceInteractions as It, ScoreboardRenderOptions as J, CrossSurfaceComponent as Jt, PlaybackDriver as K, CrossSurfaceCandidateOutcome as Kt, ProfileMatrixCoverage as L, AnalyzeCrossSurfaceInteractionsInput as Lt, resolveRunDir as M, discoverEvalFixtures as Mt, tangleTracesRoot as N, loadEvalFixture as Nt, DiscriminationScore as O, EvalFixtureScenario as Ot, CreateProfileMatrixPlanOptions as P, loadEvalFixtureScenarios as Pt, UserStoryVerdict as Q, CrossSurfaceEligibility as Qt, ProfileMatrixPlan as R, CrossSurfaceAdditionDecision as Rt, SingleRunLockOptions as S, SearchLedgerError as Sn, NeutralizationGateOptions as St, campaignBreakdown as T, EvalFixtureFile as Tt, finalizeProfileMatrix as U, CrossSurfaceCandidate as Ut, RunProfileMatrixSegmentOptions as V, CrossSurfaceBestSingleSelection as Vt, runProfileMatrixSegment as W, CrossSurfaceCandidateComparison as Wt, ScoreboardSummary as X, CrossSurfaceCompositionStep as Xt, ScoreboardRow as Y, CrossSurfaceComponentEvidence as Yt, UserStory as Z, CrossSurfaceDistribution as Zt, componentSurfaceIdentityMaterial as _, TraceAnalystArtifact as _n, RolloutCall as _t, WorktreeAdapterError as a, CrossSurfaceInteractionTask as an, ProfileMatrixError as at, surfaceHash as b, traceAnalystQualityJudge as bn, classifyUngroundedLiterals as bt, verifyCodeSurface as c, CrossSurfacePairEvidence as cn, RunProfileMatrixResult as ct, PhoenixEvaluationResultLike as d, CrossSurfaceRankedSingle as dn, neutralizeText as dt, CrossSurfaceIneligibilityReason as en, renderScoreboardMarkdown as et, PhoenixEvaluatorLike as f, CrossSurfaceRelativeCost as fn, FsLabeledScenarioStore as ft, codeSurfaceIdentityMaterial as g, BuildTraceAnalystSurfaceDispatchOptions as gn, RolloutArgumentDiffOptions as gt, assertCodeSurfaceIdentity as h, CrossSurfaceTaskRow as hn, RolloutArgumentDiff as ht, WorktreeAdapter as i, CrossSurfaceInteractionReport as in, ProfileDispatchFn as it, selectDiscriminative as j, PlanEvalFixtureRunOptions as jt, ScenarioSignal as k, EvalFixtureValidationMode as kt, AutoevalsScoreLike as l, CrossSurfacePairIncompatibilityReason as ln, ScenarioRollup as lt, phoenixEvaluatorJudge as m, CrossSurfaceSelections as mn, LabeledScenarioStoreError as mt, GitWorktreeAdapterOptions as n, CrossSurfaceInteractionEffect as nn, scoreboardSummary as nt, gitWorktreeAdapter as o, CrossSurfaceNaiveStackSelection as on, ProfileSummary as ot, autoevalsScorerJudge as p, CrossSurfaceSelectionPolicy as pn, FsLabeledScenarioStoreOptions as pt, PlaybackStep as q, CrossSurfaceCandidateSummary as qt, Worktree as r, CrossSurfaceInteractionPath as rn, userStoryScoreboard as rt, resolveWorktreePath as s, CrossSurfacePairCompatibility as sn, RunProfileMatrixOptions as st, CodeSurfaceVerification as t, CrossSurfaceInteractionAwareSelection as tn, scoreUserStory as tt, AutoevalsScorerLike as u, CrossSurfacePairwiseEntry as un, runProfileMatrix as ut, renderSurfaceDiff as v, TraceAnalystScenario as vn, ScoredRollout as vt, CampaignBreakdown as w, EvalFixture as wt, SingleRunLock as x, SearchLedgerConflictError as xn, rolloutArgumentDiff as xt, surfaceContentHash as y, buildTraceAnalystSurfaceDispatch as yn, UngroundedLiteralReport as yt, ProfileMatrixRow as z, CrossSurfaceAdditionRejectionReason as zt };
|
|
1135
|
-
//# sourceMappingURL=index-CFDffsKz.d.ts.map
|