@tangle-network/agent-eval 0.133.2 → 0.134.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +218 -0
- package/dist/analyst/index.d.ts +11 -35
- package/dist/analyst/index.d.ts.map +1 -1
- package/dist/analyst/index.js +4 -53
- package/dist/analyst/index.js.map +1 -1
- package/dist/{analyze-runs-BmX-h_yn.d.ts → analyze-runs-DMo3Lb_y.d.ts} +4 -4
- package/dist/{analyze-runs-BmX-h_yn.d.ts.map → analyze-runs-DMo3Lb_y.d.ts.map} +1 -1
- package/dist/{analyze-runs-B-afTpCv.js → analyze-runs-qk8op0tN.js} +63 -42
- package/dist/analyze-runs-qk8op0tN.js.map +1 -0
- package/dist/baseline-BaPxoROc.js +149 -0
- package/dist/baseline-BaPxoROc.js.map +1 -0
- package/dist/{baseline-hG3K85h4.d.ts → baseline-D_fT6277.d.ts} +43 -11
- package/dist/baseline-D_fT6277.d.ts.map +1 -0
- package/dist/benchmarks/index.d.ts +1 -1
- package/dist/benchmarks/index.js +1 -1
- package/dist/{benchmarks-CJr1H1_a.js → benchmarks-v5piCeDl.js} +3 -3
- package/dist/{benchmarks-CJr1H1_a.js.map → benchmarks-v5piCeDl.js.map} +1 -1
- package/dist/builder-eval/index.js +1 -1
- package/dist/campaign/index.d.ts +5 -4
- package/dist/campaign/index.js +4 -3
- package/dist/{campaign-BJjn1rhw.js → campaign-DEC_7DLn.js} +12 -6
- package/dist/{campaign-BJjn1rhw.js.map → campaign-DEC_7DLn.js.map} +1 -1
- package/dist/{client-COvaLoQG.d.ts → client-BIyh1RCr.d.ts} +29 -15
- package/dist/client-BIyh1RCr.d.ts.map +1 -0
- package/dist/{client-CYzbdJOZ.js → client-LIuo-KPv.js} +19 -7
- package/dist/client-LIuo-KPv.js.map +1 -0
- package/dist/contract/index.d.ts +12 -11
- package/dist/contract/index.d.ts.map +1 -1
- package/dist/contract/index.js +12 -18
- package/dist/contract/index.js.map +1 -1
- package/dist/{default-registry-Cl3pHo4n.d.ts → default-registry-Brxr728w.d.ts} +4 -268
- package/dist/default-registry-Brxr728w.d.ts.map +1 -0
- package/dist/{default-registry-D3T9XbuY.js → default-registry-IjYs7T8l.js} +4 -61
- package/dist/default-registry-IjYs7T8l.js.map +1 -0
- package/dist/{eval-campaign-DXhpZghy.js → eval-campaign-CvPcvqXC.js} +2 -2
- package/dist/{eval-campaign-DXhpZghy.js.map → eval-campaign-CvPcvqXC.js.map} +1 -1
- package/dist/hosted/index.d.ts +2 -2
- package/dist/hosted/index.d.ts.map +1 -1
- package/dist/hosted/index.js +1 -1
- package/dist/{index-C7Wue8R6.d.ts → index-BoJNQR6n.d.ts} +29 -11
- package/dist/index-BoJNQR6n.d.ts.map +1 -0
- package/dist/{index-BREtv3ZZ.d.ts → index-C21xKtxu.d.ts} +4 -4
- package/dist/{index-BREtv3ZZ.d.ts.map → index-C21xKtxu.d.ts.map} +1 -1
- package/dist/{index-DSC51roc.d.ts → index-DSC51roc2.d.ts} +1 -1
- package/dist/index-DSC51roc2.d.ts.map +1 -0
- package/dist/{index-nhIYz9hn.d.ts → index-DuhJaaiH.d.ts} +68 -7
- package/dist/index-DuhJaaiH.d.ts.map +1 -0
- package/dist/index.d.ts +60 -13
- package/dist/index.d.ts.map +1 -1
- package/dist/index.js +134 -27
- package/dist/index.js.map +1 -1
- package/dist/ledger-core/index.d.ts +2 -2
- package/dist/ledger-core/index.js +2 -2
- package/dist/{ledger-core-CPZfcrC2.js → ledger-core-DAKFKRzi.js} +136 -18
- package/dist/ledger-core-DAKFKRzi.js.map +1 -0
- package/dist/matrix/index.d.ts +1 -1
- package/dist/meta-eval/index.d.ts +1 -1
- package/dist/meta-eval/index.js +2 -2
- package/dist/multishot/index.d.ts +2 -2
- package/dist/openapi.json +1 -1
- package/dist/{paired-arms-6XItKzd1.js → paired-arms-CA_8pN01.js} +2 -2
- package/dist/{paired-arms-6XItKzd1.js.map → paired-arms-CA_8pN01.js.map} +1 -1
- package/dist/pipelines/index.d.ts +1 -1
- package/dist/pipelines/index.js +3 -2
- package/dist/pipelines/index.js.map +1 -1
- package/dist/proposal-findings-DCawte-y.js +164 -0
- package/dist/proposal-findings-DCawte-y.js.map +1 -0
- package/dist/{release-report-wuilQkvK.js → release-report-BVZBmRZp.js} +2 -2
- package/dist/{release-report-wuilQkvK.js.map → release-report-BVZBmRZp.js.map} +1 -1
- package/dist/{release-report-CjHWa8Ia.d.ts → release-report-CuULWKyk.d.ts} +2 -2
- package/dist/{release-report-CjHWa8Ia.d.ts.map → release-report-CuULWKyk.d.ts.map} +1 -1
- package/dist/reporting.d.ts +3 -3
- package/dist/reporting.js +4 -4
- package/dist/{researcher-CbSKhK8z.d.ts → researcher-DVtruQ9U.d.ts} +2 -2
- package/dist/{researcher-CbSKhK8z.d.ts.map → researcher-DVtruQ9U.d.ts.map} +1 -1
- package/dist/{reward-hacking-Dl2UBzej.js → reward-hacking-DCdRK9TY.js} +2 -2
- package/dist/{reward-hacking-Dl2UBzej.js.map → reward-hacking-DCdRK9TY.js.map} +1 -1
- package/dist/rl.d.ts +2 -2
- package/dist/rl.d.ts.map +1 -1
- package/dist/rl.js +18 -5
- package/dist/rl.js.map +1 -1
- package/dist/{rubric-predictive-validity-QG7ydk0s.js → rubric-predictive-validity-D6Q6n9oq.js} +2 -2
- package/dist/{rubric-predictive-validity-QG7ydk0s.js.map → rubric-predictive-validity-D6Q6n9oq.js.map} +1 -1
- package/dist/{semantic-concept-judge-BypLt6Fw.js → semantic-concept-judge-C0P1VTXD.js} +2 -3
- package/dist/{semantic-concept-judge-BypLt6Fw.js.map → semantic-concept-judge-C0P1VTXD.js.map} +1 -1
- package/dist/{skill-usage-BaaxFSJR.d.ts → skill-usage-BDQVPIG1.d.ts} +3 -2
- package/dist/skill-usage-BDQVPIG1.d.ts.map +1 -0
- package/dist/{skillopt-optimization-method-CF6a327Q.js → skillopt-optimization-method-BY6vKLJB.js} +169 -53
- package/dist/skillopt-optimization-method-BY6vKLJB.js.map +1 -0
- package/dist/{skillopt-optimization-method-wHF5xsUv.d.ts → skillopt-optimization-method-DJ3l4w8W.d.ts} +20 -18
- package/dist/skillopt-optimization-method-DJ3l4w8W.d.ts.map +1 -0
- package/dist/{statistics-DbvkkDPa.d.ts → statistics-D_4Snl-5.d.ts} +158 -30
- package/dist/statistics-D_4Snl-5.d.ts.map +1 -0
- package/dist/{statistics-DWM_AyLe.js → statistics-RwRNu2__.js} +546 -98
- package/dist/statistics-RwRNu2__.js.map +1 -0
- package/dist/{summary-report-Ci17nIdU.js → summary-report-BxtossFi.js} +3 -3
- package/dist/{summary-report-Ci17nIdU.js.map → summary-report-BxtossFi.js.map} +1 -1
- package/dist/{summary-report-CFnQgNfg.d.ts → summary-report-DGp0-_XO.d.ts} +61 -4
- package/dist/summary-report-DGp0-_XO.d.ts.map +1 -0
- package/dist/{baseline-DcX5hQDv.js → tool-use-metrics-DEGMKycK.js} +2 -114
- package/dist/tool-use-metrics-DEGMKycK.js.map +1 -0
- package/dist/types-DVjczBM9.d.ts +276 -0
- package/dist/types-DVjczBM9.d.ts.map +1 -0
- package/dist/{types-BokuXvOG.d.ts → types-DiWLru6Z.d.ts} +20 -37
- package/dist/types-DiWLru6Z.d.ts.map +1 -0
- package/docs/campaign-proposers.md +5 -0
- package/docs/design/statistics-decisions.md +271 -0
- package/docs/design.md +1 -0
- package/docs/insight-report.md +1 -1
- package/docs/research-report-methodology.md +4 -1
- package/package.json +2 -1
- package/dist/analyze-runs-B-afTpCv.js.map +0 -1
- package/dist/baseline-DcX5hQDv.js.map +0 -1
- package/dist/baseline-hG3K85h4.d.ts.map +0 -1
- package/dist/client-COvaLoQG.d.ts.map +0 -1
- package/dist/client-CYzbdJOZ.js.map +0 -1
- package/dist/default-registry-Cl3pHo4n.d.ts.map +0 -1
- package/dist/default-registry-D3T9XbuY.js.map +0 -1
- package/dist/index-C7Wue8R6.d.ts.map +0 -1
- package/dist/index-DSC51roc.d.ts.map +0 -1
- package/dist/index-nhIYz9hn.d.ts.map +0 -1
- package/dist/ledger-core-CPZfcrC2.js.map +0 -1
- package/dist/run-score-iEEAWiBY.js +0 -41
- package/dist/run-score-iEEAWiBY.js.map +0 -1
- package/dist/skill-usage-BaaxFSJR.d.ts.map +0 -1
- package/dist/skillopt-optimization-method-CF6a327Q.js.map +0 -1
- package/dist/skillopt-optimization-method-wHF5xsUv.d.ts.map +0 -1
- package/dist/statistics-DWM_AyLe.js.map +0 -1
- package/dist/statistics-DbvkkDPa.d.ts.map +0 -1
- package/dist/summary-report-CFnQgNfg.d.ts.map +0 -1
- package/dist/types-BokuXvOG.d.ts.map +0 -1
|
@@ -1,8 +1,8 @@
|
|
|
1
1
|
import { o as FailureClass } from "./schema-BtVldJ3T.js";
|
|
2
2
|
import { l as RunTerminalOutcome } from "./run-record-DcObtIGh.js";
|
|
3
|
-
import { _ as GateDecision, j as MutableSurface } from "./types-
|
|
3
|
+
import { _ as GateDecision, j as MutableSurface } from "./types-DiWLru6Z.js";
|
|
4
4
|
import { r as ContinuousAgreement } from "./judge-calibration-DFtEMlde.js";
|
|
5
|
-
import { i as ParetoFigureSpec, t as GainDistributionBin } from "./summary-report-
|
|
5
|
+
import { i as ParetoFigureSpec, t as GainDistributionBin } from "./summary-report-DGp0-_XO.js";
|
|
6
6
|
//#region src/contract/insight-report.d.ts
|
|
7
7
|
interface InsightReport {
|
|
8
8
|
/** Number of runs analyzed. */
|
|
@@ -249,10 +249,15 @@ interface LiftInsight {
|
|
|
249
249
|
delta: number;
|
|
250
250
|
/** Lower / upper bound of bootstrap CI on the delta. */
|
|
251
251
|
ci95: [number, number];
|
|
252
|
-
/** Paired-t-test p-value
|
|
253
|
-
|
|
252
|
+
/** Paired-t-test p-value; null when the delta is a non-zero constant, where
|
|
253
|
+
* the t statistic is undefined. */
|
|
254
|
+
pValue: number | null;
|
|
254
255
|
/** Number of paired observations. */
|
|
255
256
|
n: number;
|
|
257
|
+
/** Minimum paired observations required before the interval can drive a decision. */
|
|
258
|
+
minimumRequired: number;
|
|
259
|
+
/** Whether the bootstrap interval has enough observations to drive a decision. */
|
|
260
|
+
decisionEligible: boolean;
|
|
256
261
|
/** Scored baseline observations without a candidate match. */
|
|
257
262
|
unpairedBaseline: number;
|
|
258
263
|
/** Scored candidate observations without a baseline match. */
|
|
@@ -331,7 +336,7 @@ interface ReleaseSummary {
|
|
|
331
336
|
* consumers can post-process to populate. */
|
|
332
337
|
issues: string[];
|
|
333
338
|
}
|
|
334
|
-
interface
|
|
339
|
+
interface MetricDeltaBase {
|
|
335
340
|
/** Current-period mean. */
|
|
336
341
|
current: number;
|
|
337
342
|
/** Baseline-period mean. */
|
|
@@ -340,21 +345,28 @@ interface MetricDelta {
|
|
|
340
345
|
* the consumer-side interpretation: "higher current" — semantic
|
|
341
346
|
* direction depends on the metric). */
|
|
342
347
|
delta: number;
|
|
343
|
-
/**
|
|
344
|
-
|
|
348
|
+
/** Sample sizes. */
|
|
349
|
+
baselineN: number;
|
|
350
|
+
currentN: number;
|
|
351
|
+
}
|
|
352
|
+
type MetricDelta = (MetricDeltaBase & {
|
|
353
|
+
status: 'ok';
|
|
354
|
+
/** Welch 95% confidence interval on the delta. Two-sample, unpaired. */
|
|
345
355
|
ci95: [number, number];
|
|
346
356
|
/** Welch t-test p-value (two-sided). */
|
|
347
357
|
pValue: number;
|
|
348
358
|
/** Cohen's d (pooled stddev). Effect size, signed. */
|
|
349
359
|
cohensD: number;
|
|
350
|
-
/**
|
|
351
|
-
baselineN: number;
|
|
352
|
-
currentN: number;
|
|
353
|
-
/** True when p < 0.05 AND |d| >= 0.2 (small-effect threshold). The
|
|
354
|
-
* conjunction prevents large-effect-but-noisy and significant-but-
|
|
355
|
-
* tiny from triggering recommendations. */
|
|
360
|
+
/** True when p < 0.05 AND |d| >= 0.2. */
|
|
356
361
|
significant: boolean;
|
|
357
|
-
}
|
|
362
|
+
}) | (MetricDeltaBase & {
|
|
363
|
+
status: 'insufficient-sample' | 'zero-variance';
|
|
364
|
+
/** Null because the observed data cannot define Welch inference. */
|
|
365
|
+
ci95: null;
|
|
366
|
+
pValue: null;
|
|
367
|
+
cohensD: null;
|
|
368
|
+
significant: false;
|
|
369
|
+
});
|
|
358
370
|
interface PriorPeriodComparison {
|
|
359
371
|
/** Sample counts. */
|
|
360
372
|
baselineN: number;
|
|
@@ -370,6 +382,8 @@ interface PriorPeriodComparison {
|
|
|
370
382
|
regressedMetrics: string[];
|
|
371
383
|
/** Metric names where current is significantly BETTER than baseline. */
|
|
372
384
|
improvedMetrics: string[];
|
|
385
|
+
/** Metrics whose samples cannot define a Welch comparison. */
|
|
386
|
+
inconclusiveMetrics: string[];
|
|
373
387
|
}
|
|
374
388
|
interface Recommendation {
|
|
375
389
|
priority: 'critical' | 'high' | 'medium' | 'low';
|
|
@@ -578,4 +592,4 @@ declare function hostedClientFromEnv(overrides?: Partial<HostedTenant> & {
|
|
|
578
592
|
}): HostedClient | undefined;
|
|
579
593
|
//#endregion
|
|
580
594
|
export { ScalarDistribution as A, InsightReport as C, OutcomeCorrelationInsight as D, LiftInsight as E, Recommendation as O, FailureClusterInsight as S, JudgeInsight as T, UnixNanoTimestamp as _, hostedTenantFromEnv as a, ExecutionInsight as b, EvalRunGenerationSnapshot as c, HostedIngestHeaders as d, HostedWireVersion as f, TraceSpanEvent as g, IngestTracesRequest as h, hostedClientFromEnv as i, TokenUsageInsight as j, ReleaseSummary as k, EvalRunStatus as l, IngestResponse as m, HostedTenant as n, EvalRunCellScore as o, IngestEvalRunsRequest as p, createHostedClient as r, EvalRunEvent as s, HostedClient as t, HOSTED_WIRE_VERSION as u, CostProvenanceSummary as v, InterRaterInsight as w, FailureClassTally as x, ExecutionErrorOutcomeCell as y };
|
|
581
|
-
//# sourceMappingURL=client-
|
|
595
|
+
//# sourceMappingURL=client-BIyh1RCr.d.ts.map
|
|
@@ -0,0 +1 @@
|
|
|
1
|
+
{"version":3,"file":"client-BIyh1RCr.d.ts","names":[],"sources":["../src/contract/insight-report.ts","../src/hosted/types.ts","../src/hosted/client.ts"],"mappings":";;;;;;UAqCiB;;EAEf;;;;EAKA,WAAW;;EAGX,WAAW;;;EAIX,cAAc,eAAe;;EAG7B;IACE,MAAM;IACN,QAAQ;;;;IAIR,aAAa;;;;;IAKb;MAAa;MAAe;;;;;;EAM9B,QAAQ,eAAe;;;;EAKvB,aAAa;;;;EAKb,OAAO;;;EAIP,kBAAkB;;;EAIlB,gBAAgB;;;;;EAMhB,qBAAqB;;;;;EAMrB,SAAS;;;;;;EAOT,wBAAwB;;;;EAKxB,iBAAiB;;;EAIjB,iBAAiB;;UAGF;EACf;IAAY;IAAW;;EACvB;IAAa;IAAW;;EACxB;IAAc;;EACd;;UAGe;;EAEf,YAAY;;EAEZ,SAAS;;;EAGT,YAAY;;;EAGZ;IACE;IACA,YAAY;IACZ,SAAS;IACT;;;EAGF,QAAQ;IAAQ;IAAe;;;;;EAI/B;IACE;IACA;IACA;;;;;EAKF;IACE;;;IAGA;;IAEA;;IAEA;;IAEA;;IAEA;;;;;IAKA,mBAAmB,OAAO,oBAAoB;;;;EAIhD;IACE;IACA;IACA;IACA;IACA;;;UAIa;;EAEf;;EAEA;;EAEA;;UAGe;EACf,OAAO;EACP,QAAQ;EACR,WAAW;EACX,QAAQ;EACR,YAAY;EACZ;IACE;IACA;IACA;IACA;IACA;;;;UAOa;;EAEf;;EAEA;;EAEA;;EAEA;;EAEA;;EAEA;;EAEA;;EAEA,WAAW;;;;;EAKX,WAAW;IAAQ;IAAe;;;UAGnB;;EAEf;;EAEA;;;EAGA,cAAc;;;EAGd;;;EAGA;;;EAGA;;UAGe;;EAEf;;EAEA;;EAEA;;EAEA;;EAEA;;EAEA;;EAEA,SAAS;;EAET,mBAAmB;IACjB;IACA,SAAS;MAAQ;MAAe;;IAChC;;;UAIa;EACf;EACA;;EAEA;;EAEA;;;EAGA;;EAEA;;EAEA;;EAEA;;EAEA;;EAEA;;EAEA;;EAEA;;EAEA;;UAGe;;EAEf,UAAU;IACR;IACA;;IAEA;;IAEA;;IAEA;;EAEF;;;;;UAMe;;EAEf,cAAc;;EAEd;;EAEA;;UAGe;;EAEf;;;EAGA;EACA,UAAU;IAAQ;IAAe;IAAgB;;;UAGlC;;;EAGf;;EAEA;;EAEA;;EAEA;;EAEA;IACE;IACA;IACA;;;UAIa;;;EAGf;EACA,MAAM;IACJ;IACA;IACA;;;;EAIF;;UAGQ;;EAER;;EAEA;;;;EAIA;;EAEA;EACA;;KAGU,eACP;EACC;;EAEA;;EAEA;;EAEA;;EAEA;MAED;EACC;;EAEA;EACA;EACA;EACA;;UAGW;;EAEf;EACA;;EAEA;;;;EAIA,SAAS,eAAe;;;EAGxB;;EAEA;;EAEA;;UAGe;EACf;EACA;EACA;EACA;;EAEA;;;;cCnZW;KACD,2BAA2B;;UAKtB;;EAEf;;EAEA;;EAEA,yBAAyB;;EAEzB;;;KAMU;UAQK;;EAEf;;EAEA;;EAEA;;EAEA,YAAY,eAAe;;EAE3B,iBAAiB;;EAEjB;;EAEA;;UAGe;;EAEf;;;EAGA;;;EAGA,UAAU;;EAEV,OAAO;;EAEP;;EAEA;;EAEA;;;;;;;UAQe;;EAEf;;EAEA;;EAEA;;EAEA,QAAQ;;EAER,QAAQ;;EAER,WAAW;;EAEX,aAAa;;EAEb,eAAe;;EAEf;;EAEA;;EAEA;;EAEA;;;;;;EAMA,gBAAgB;;;;;;KASN;;;;;;UAOK;EACf;EACA;EACA;EACA;EACA,mBAAmB;EACnB,iBAAiB;EACjB,YAAY;EACZ,SAAS;IACP,cAAc;IACd;IACA,aAAa;;EAEf;IAAW;IAAgC;;;EAE3C;;EAEA;;EAEA;;EAEA;;UAKe;EACf,aAAa;EACb,QAAQ;;UAGO;EACf,aAAa;EACb,OAAO;;UAGQ;;EAEf;;EAEA,UAAU;IAAQ;IAAe;;;;;UC1JlB;;EAEf;;EAEA;;EAEA;;EAEA,mBAAmB;;EAEnB;;EAEA;;UAGe;EACf,cAAc,OAAO,cAAc,0BAA0B,QAAQ;EACrE,eAAe,QAAQ,gBAAgB,0BAA0B,QAAQ;EACzE,aAAa,OAAO,kBAAkB,0BAA0B,QAAQ;WAC/D,QAAQ;WACR,aAAa;;iBAyGR,mBAAmB,QAAQ,eAAe;;;;;;;;;;;;;;;;;;;;;;;;;;;iBAsE1C,oBACd,YAAW,QAAQ;EAAkB,MAAM;IAC1C;iBAiBa,oBACd,YAAW,QAAQ;EAAkB,MAAM;IAC1C"}
|
|
@@ -189,8 +189,10 @@ const LiftInsightSchema = z.object({
|
|
|
189
189
|
candidateMean: finiteNumber,
|
|
190
190
|
delta: finiteNumber,
|
|
191
191
|
ci95: z.tuple([finiteNumber, finiteNumber]),
|
|
192
|
-
pValue: finiteNumber.min(0).max(1),
|
|
192
|
+
pValue: finiteNumber.min(0).max(1).nullable(),
|
|
193
193
|
n: nonNegativeInteger,
|
|
194
|
+
minimumRequired: z.number().int().positive(),
|
|
195
|
+
decisionEligible: z.boolean(),
|
|
194
196
|
unpairedBaseline: nonNegativeInteger,
|
|
195
197
|
unpairedCandidate: nonNegativeInteger,
|
|
196
198
|
cohensD: finiteNumber.nullable(),
|
|
@@ -249,24 +251,34 @@ const ReleaseSummarySchema = z.object({
|
|
|
249
251
|
}).strict()),
|
|
250
252
|
issues: z.array(z.string())
|
|
251
253
|
}).strict();
|
|
252
|
-
const
|
|
254
|
+
const MetricDeltaBaseSchema = z.object({
|
|
253
255
|
current: finiteNumber,
|
|
254
256
|
baseline: finiteNumber,
|
|
255
257
|
delta: finiteNumber,
|
|
258
|
+
baselineN: nonNegativeInteger,
|
|
259
|
+
currentN: nonNegativeInteger
|
|
260
|
+
}).strict();
|
|
261
|
+
const MetricDeltaSchema = z.discriminatedUnion("status", [MetricDeltaBaseSchema.extend({
|
|
262
|
+
status: z.literal("ok"),
|
|
256
263
|
ci95: z.tuple([finiteNumber, finiteNumber]),
|
|
257
264
|
pValue: finiteNumber.min(0).max(1),
|
|
258
265
|
cohensD: finiteNumber,
|
|
259
|
-
baselineN: nonNegativeInteger,
|
|
260
|
-
currentN: nonNegativeInteger,
|
|
261
266
|
significant: z.boolean()
|
|
262
|
-
}).
|
|
267
|
+
}), MetricDeltaBaseSchema.extend({
|
|
268
|
+
status: z.enum(["insufficient-sample", "zero-variance"]),
|
|
269
|
+
ci95: z.null(),
|
|
270
|
+
pValue: z.null(),
|
|
271
|
+
cohensD: z.null(),
|
|
272
|
+
significant: z.literal(false)
|
|
273
|
+
})]);
|
|
263
274
|
const PriorPeriodComparisonSchema = z.object({
|
|
264
275
|
baselineN: nonNegativeInteger,
|
|
265
276
|
currentN: nonNegativeInteger,
|
|
266
277
|
windowLabel: nonEmptyString.optional(),
|
|
267
278
|
metrics: z.record(z.string(), MetricDeltaSchema),
|
|
268
279
|
regressedMetrics: z.array(nonEmptyString),
|
|
269
|
-
improvedMetrics: z.array(nonEmptyString)
|
|
280
|
+
improvedMetrics: z.array(nonEmptyString),
|
|
281
|
+
inconclusiveMetrics: z.array(nonEmptyString)
|
|
270
282
|
}).strict();
|
|
271
283
|
const RecommendationSchema = z.object({
|
|
272
284
|
priority: z.enum([
|
|
@@ -634,4 +646,4 @@ function hostedClientFromEnv(overrides = {}) {
|
|
|
634
646
|
//#endregion
|
|
635
647
|
export { EvalRunEventSchema as a, IngestResponseSchema as c, MutableSurfaceSchema as d, RunTerminalOutcomeSchema as f, HOSTED_WIRE_VERSION as h, EvalRunCellScoreSchema as i, IngestTracesRequestSchema as l, UnixNanoTimestampSchema as m, hostedClientFromEnv as n, EvalRunGenerationSnapshotSchema as o, TraceSpanEventSchema as p, hostedTenantFromEnv as r, IngestEvalRunsRequestSchema as s, createHostedClient as t, InsightReportSchema as u };
|
|
636
648
|
|
|
637
|
-
//# sourceMappingURL=client-
|
|
649
|
+
//# sourceMappingURL=client-LIuo-KPv.js.map
|
|
@@ -0,0 +1 @@
|
|
|
1
|
+
{"version":3,"file":"client-LIuo-KPv.js","names":[],"sources":["../src/hosted/types.ts","../src/hosted/schemas.ts","../src/hosted/client.ts"],"sourcesContent":["/**\n * # Hosted-tier wire format — the schema that EVERY orchestrator (ours,\n * a partner's self-hosted one, a future open implementation) must accept.\n *\n * This package implements exactly one wire version. Servers reject every\n * other version instead of translating old payloads.\n *\n * The wire format is two event streams in one transport:\n *\n * 1. **Eval-run events** (`POST /v1/ingest/eval-runs`). Posted when a\n * campaign / improvement-loop completes (or per-generation if\n * streaming). Carries the structured result + per-cell scores +\n * surface diffs the orchestrator stores for the dashboard.\n *\n * 2. **Trace spans** (`POST /v1/ingest/traces`). Standard OTLP-shaped\n * spans with a few additional attributes so the orchestrator can\n * pivot from eval-run → underlying execution. Compatible with any\n * OTel collector.\n *\n * Both endpoints are authenticated with a bearer token + a tenant id\n * header. Tenants isolate everything downstream of ingest; no tenant\n * ever sees another tenant's data.\n */\n\nimport type { GateDecision, MutableSurface } from '../campaign/types'\nimport type { InsightReport } from '../contract/insight-report'\nimport type { RunTerminalOutcome } from '../run-record'\n\n// re-export so wire-format consumers can import the optional payload type\n// from `@tangle-network/agent-eval/hosted` without reaching into /contract.\nexport type { InsightReport } from '../contract/insight-report'\n\nexport const HOSTED_WIRE_VERSION = '2026-07-24.v1' as const\nexport type HostedWireVersion = typeof HOSTED_WIRE_VERSION\n\n// ── Transport headers ───────────────────────────────────────────────\n\n/** Every ingest request carries these. */\nexport interface HostedIngestHeaders {\n /** Bearer token. The orchestrator validates against the tenant key. */\n authorization: `Bearer ${string}`\n /** Stable tenant id (the orchestrator-side primary key for the tenant). */\n 'x-tangle-tenant-id': string\n /** Wire-version pin so the server can reject incompatible payloads. */\n 'x-tangle-wire-version': HostedWireVersion\n /** Stable request key generated once and reused across retries. */\n 'idempotency-key': string\n}\n\n// ── Eval-run event ──────────────────────────────────────────────────\n\n/** Lifecycle stages of an eval-run as the substrate reports them. */\nexport type EvalRunStatus =\n | 'started'\n | 'baseline-complete'\n | 'generation-complete'\n | 'gate-decided'\n | 'finished'\n | 'errored'\n\nexport interface EvalRunCellScore {\n /** Stable scenario id from the consumer's scenario set. */\n scenarioId: string\n /** Repetition index when reps > 1; 0 for the default. */\n rep: number\n /** Composite score across successful judges, or null when unscored. */\n compositeMean: number | null\n /** Per-judge and per-dimension scores; failed or missing judges are absent. */\n dimensions: Record<string, Record<string, number>>\n /** Root execution result, kept separate from task quality. */\n terminalOutcome: RunTerminalOutcome\n /** Canonical execution-error count, or null when the producer did not measure it. */\n executionErrorCount: number | null\n /** Per-cell dispatch or judge error. Missing on success. */\n errorMessage?: string\n}\n\nexport interface EvalRunGenerationSnapshot {\n /** Generation index. 0 is baseline. */\n index: number\n /** Candidate surface fingerprint (stable hash) — pivot key into the\n * trace stream to fetch the underlying execution. */\n surfaceHash: string\n /** The candidate surface itself. May be omitted to avoid PII when the\n * consumer prefers not to ship verbatim prompts. */\n surface?: MutableSurface\n /** Per-cell scores for this generation. */\n cells: EvalRunCellScore[]\n /** Mean across scored cells, or null when no cell has a task-quality label. */\n compositeMean: number | null\n /** Total $ spent across this generation. */\n costUsd: number\n /** Wall-clock duration of this generation. */\n durationMs: number\n}\n\n/**\n * The top-level eval-run event. One ingest call per logical eval-run;\n * generations stream in incrementally via repeated calls with the same\n * `runId`. The orchestrator deduplicates by `(runId, generation.index)`.\n */\nexport interface EvalRunEvent {\n /** Stable run id (the substrate's `runId`). UUID or substrate-generated. */\n runId: string\n /** Where this run was happening — derived from `RunCampaignOptions.runDir`. */\n runDir: string\n /** ISO-8601 timestamp the substrate recorded the event. */\n timestamp: string\n /** Lifecycle stage this event represents. */\n status: EvalRunStatus\n /** Free-form consumer tags (env, branch, model id, etc.). Searchable. */\n labels: Record<string, string>\n /** Baseline campaign snapshot. Present when status >= baseline-complete. */\n baseline?: EvalRunGenerationSnapshot\n /** Per-generation snapshots. Streams in; orchestrator appends. */\n generations: EvalRunGenerationSnapshot[]\n /** Final gate decision. Present when status >= gate-decided. */\n gateDecision?: GateDecision\n /** Held-out lift = winner-on-holdout - baseline-on-holdout. */\n holdoutLift?: number\n /** Total $ spent across baseline + every generation. */\n totalCostUsd: number\n /** Total wall-clock duration. */\n totalDurationMs: number\n /** Error message if status === 'errored'. */\n errorMessage?: string\n /** Rigor packet emitted alongside the run — distributional summary,\n * paired-bootstrap lift CI, judge stats, inter-rater agreement,\n * contamination check, failure clusters (when an analyst is wired),\n * outcome correlation (when downstream signal is supplied), and the\n * recommendations the dashboard surfaces verbatim. */\n insightReport?: InsightReport\n}\n\n// ── Trace span event ────────────────────────────────────────────────\n\n/**\n * Canonical unsigned 64-bit integer encoded as a base-10 string.\n * JSON numbers cannot represent OTLP nanosecond timestamps exactly.\n */\nexport type UnixNanoTimestamp = string\n\n/**\n * OTel-shape span with a few additional attributes for eval-run pivoting.\n * Compatible with any OTLP collector — `name`, `traceId`, `spanId`,\n * `startTimeUnixNano`, `endTimeUnixNano`, `attributes` are stock OTel.\n */\nexport interface TraceSpanEvent {\n traceId: string\n spanId: string\n parentSpanId?: string\n name: string\n startTimeUnixNano: UnixNanoTimestamp\n endTimeUnixNano: UnixNanoTimestamp\n attributes: Record<string, string | number | boolean>\n events?: Array<{\n timeUnixNano: UnixNanoTimestamp\n name: string\n attributes?: Record<string, string | number | boolean>\n }>\n status?: { code: 'OK' | 'ERROR' | 'UNSET'; message?: string }\n /** Pivot back into the eval-run stream. */\n 'tangle.runId'?: string\n /** Pivot to the specific generation. */\n 'tangle.generation'?: number\n /** Pivot to the specific cell. */\n 'tangle.cellId'?: string\n /** Pivot to the specific scenario. */\n 'tangle.scenarioId'?: string\n}\n\n// ── Ingest request bodies ───────────────────────────────────────────\n\nexport interface IngestEvalRunsRequest {\n wireVersion: HostedWireVersion\n events: EvalRunEvent[]\n}\n\nexport interface IngestTracesRequest {\n wireVersion: HostedWireVersion\n spans: TraceSpanEvent[]\n}\n\nexport interface IngestResponse {\n /** Accepted events / spans count. */\n accepted: number\n /** Rejected events with reasons (validation failures, dup idempotency key, etc.). */\n rejected: Array<{ index: number; reason: string }>\n}\n","import { z } from 'zod'\nimport type { MutableSurface } from '../campaign/types'\nimport type { InsightReport } from '../contract/insight-report'\nimport { FAILURE_CLASSES, type FailureClass } from '../trace/schema'\nimport type {\n EvalRunCellScore,\n EvalRunEvent,\n EvalRunGenerationSnapshot,\n IngestEvalRunsRequest,\n IngestResponse,\n IngestTracesRequest,\n TraceSpanEvent,\n UnixNanoTimestamp,\n} from './types'\nimport { HOSTED_WIRE_VERSION } from './types'\n\nconst finiteNumber = z.number().finite()\nconst nonNegativeNumber = finiteNumber.nonnegative()\nconst nonNegativeInteger = z.number().int().nonnegative()\nconst nonEmptyString = z\n .string()\n .min(1)\n .refine((value) => value.trim().length > 0, 'must not be blank')\nconst attributeValue = z.union([z.string(), finiteNumber, z.boolean()])\nconst attributes = z.record(z.string(), attributeValue)\nconst FailureClassSchema = z.custom<FailureClass>(\n (value) => typeof value === 'string' && FAILURE_CLASSES.includes(value as FailureClass),\n `expected one of ${FAILURE_CLASSES.join(', ')}`,\n)\nconst UINT64_MAX = 18_446_744_073_709_551_615n\n\nexport const UnixNanoTimestampSchema: z.ZodType<UnixNanoTimestamp> = z\n .string()\n .regex(/^(0|[1-9][0-9]*)$/, 'expected an unsigned base-10 integer string')\n .refine((value) => BigInt(value) <= UINT64_MAX, 'must fit in an unsigned 64-bit integer')\n\nconst GainDistributionBinSchema = z\n .object({\n lo: finiteNumber,\n hi: finiteNumber,\n count: nonNegativeInteger,\n })\n .strict()\n\nconst ScalarDistributionSchema = z\n .object({\n n: nonNegativeInteger,\n mean: finiteNumber.nullable(),\n p50: finiteNumber.nullable(),\n p95: finiteNumber.nullable(),\n stddev: nonNegativeNumber.nullable(),\n min: finiteNumber.nullable(),\n max: finiteNumber.nullable(),\n histogram: z.array(GainDistributionBinSchema),\n tailRuns: z\n .array(\n z\n .object({\n runId: nonEmptyString,\n score: finiteNumber,\n })\n .strict(),\n )\n .optional(),\n })\n .strict()\n .superRefine((distribution, ctx) => {\n const values = [\n distribution.mean,\n distribution.p50,\n distribution.p95,\n distribution.stddev,\n distribution.min,\n distribution.max,\n ]\n if (distribution.n === 0 && values.some((value) => value !== null)) {\n ctx.addIssue({\n code: 'custom',\n message: 'distribution values must be null when n is 0',\n })\n }\n if (distribution.n > 0 && values.some((value) => value === null)) {\n ctx.addIssue({\n code: 'custom',\n message: 'distribution values must be numbers when n is greater than 0',\n })\n }\n if (\n distribution.min !== null &&\n distribution.max !== null &&\n distribution.min > distribution.max\n ) {\n ctx.addIssue({ code: 'custom', path: ['min'], message: 'min must not exceed max' })\n }\n })\n\nconst TokenUsageInsightSchema = z\n .object({\n input: ScalarDistributionSchema,\n output: ScalarDistributionSchema,\n reasoning: ScalarDistributionSchema,\n cached: ScalarDistributionSchema,\n cacheWrite: ScalarDistributionSchema,\n totals: z\n .object({\n input: nonNegativeNumber,\n output: nonNegativeNumber,\n reasoning: nonNegativeNumber,\n cached: nonNegativeNumber,\n cacheWrite: nonNegativeNumber,\n })\n .strict(),\n })\n .strict()\n\nconst ExecutionErrorOutcomeCellSchema = z\n .object({\n withErrors: nonNegativeInteger,\n withoutErrors: nonNegativeInteger,\n unreported: nonNegativeInteger,\n })\n .strict()\n\nconst ExecutionInsightSchema = z\n .object({\n durationMs: ScalarDistributionSchema,\n queueMs: ScalarDistributionSchema,\n tokenUsage: TokenUsageInsightSchema,\n aggregateUsage: z\n .object({\n runs: nonNegativeInteger,\n tokenUsage: TokenUsageInsightSchema,\n costUsd: ScalarDistributionSchema,\n totalCostUsd: nonNegativeNumber,\n })\n .strict(),\n models: z.array(\n z\n .object({\n model: nonEmptyString,\n runs: nonNegativeInteger,\n })\n .strict(),\n ),\n modelCalls: z\n .object({\n runs: nonNegativeInteger,\n events: nonNegativeInteger,\n reportingRuns: nonNegativeInteger,\n })\n .strict(),\n executionErrors: z\n .object({\n runs: nonNegativeInteger,\n fraction: finiteNumber.min(0).max(1).nullable(),\n events: nonNegativeInteger,\n reportingRuns: nonNegativeInteger,\n errorSpanEvents: nonNegativeInteger,\n errorSpanReportingRuns: nonNegativeInteger,\n byTerminalOutcome: z\n .object({\n succeeded: ExecutionErrorOutcomeCellSchema,\n failed: ExecutionErrorOutcomeCellSchema,\n cancelled: ExecutionErrorOutcomeCellSchema,\n incomplete: ExecutionErrorOutcomeCellSchema,\n unknown: ExecutionErrorOutcomeCellSchema,\n })\n .strict(),\n })\n .strict(),\n terminalOutcomes: z\n .object({\n succeeded: nonNegativeInteger,\n failed: nonNegativeInteger,\n cancelled: nonNegativeInteger,\n incomplete: nonNegativeInteger,\n unknown: nonNegativeInteger,\n })\n .strict(),\n })\n .strict()\n\nconst CostProvenanceSummarySchema = z\n .object({\n observed: z\n .object({\n n: nonNegativeInteger,\n totalUsd: nonNegativeNumber,\n })\n .strict(),\n estimated: z\n .object({\n n: nonNegativeInteger,\n totalUsd: nonNegativeNumber,\n })\n .strict(),\n uncaptured: z.object({ n: nonNegativeInteger }).strict(),\n knownFraction: finiteNumber.min(0).max(1),\n })\n .strict()\n\nconst ParetoFigureSpecSchema = z\n .object({\n kind: z.literal('pareto-cost-quality'),\n split: z.enum(['search', 'holdout']),\n points: z.array(\n z\n .object({\n candidateId: nonEmptyString,\n cost: nonNegativeNumber,\n quality: finiteNumber,\n n: nonNegativeInteger,\n onFrontier: z.boolean(),\n gate: z.enum(['promote', 'reject']).optional(),\n })\n .strict(),\n ),\n axes: z.object({ x: z.literal('costUsd'), y: z.literal('score') }).strict(),\n })\n .strict()\n\nconst ContinuousAgreementSchema = z\n .object({\n weightedKappa: finiteNumber,\n icc: finiteNumber,\n pearson: finiteNumber,\n spearman: finiteNumber,\n ci: z\n .object({\n icc: z.tuple([finiteNumber, finiteNumber]),\n weightedKappa: z.tuple([finiteNumber, finiteNumber]),\n })\n .strict(),\n n: nonNegativeInteger,\n raters: nonNegativeInteger,\n })\n .strict()\n\nconst JudgeInsightSchema = z\n .object({\n n: nonNegativeInteger,\n meanScore: finiteNumber,\n calibration: ContinuousAgreementSchema.optional(),\n positionalBias: finiteNumber.optional(),\n selfPreference: finiteNumber.optional(),\n verbosityBias: finiteNumber.optional(),\n })\n .strict()\n\nconst InterRaterInsightSchema = z\n .object({\n raters: nonNegativeInteger,\n jointlyRated: nonNegativeInteger,\n kappa: finiteNumber,\n icc: finiteNumber,\n pearson: finiteNumber,\n spearman: finiteNumber,\n perPair: z.record(z.string(), finiteNumber),\n disagreementCases: z.array(\n z\n .object({\n runId: nonEmptyString,\n ratings: z.array(\n z\n .object({\n rater: nonEmptyString,\n score: finiteNumber,\n })\n .strict(),\n ),\n range: nonNegativeNumber,\n })\n .strict(),\n ),\n })\n .strict()\n\nconst LiftInsightSchema = z\n .object({\n baselineMean: finiteNumber,\n candidateMean: finiteNumber,\n delta: finiteNumber,\n ci95: z.tuple([finiteNumber, finiteNumber]),\n pValue: finiteNumber.min(0).max(1).nullable(),\n n: nonNegativeInteger,\n minimumRequired: z.number().int().positive(),\n decisionEligible: z.boolean(),\n unpairedBaseline: nonNegativeInteger,\n unpairedCandidate: nonNegativeInteger,\n cohensD: finiteNumber.nullable(),\n mde: nonNegativeNumber,\n requiredN: nonNegativeInteger.nullable(),\n })\n .strict()\n\nconst FailureClusterInsightSchema = z\n .object({\n clusters: z.array(\n z\n .object({\n id: nonEmptyString,\n name: nonEmptyString,\n share: finiteNumber.min(0).max(1),\n exemplars: z.array(nonEmptyString).max(5),\n suggestedFix: nonEmptyString.optional(),\n })\n .strict(),\n ),\n totalFailures: nonNegativeInteger,\n })\n .strict()\n\nconst ContaminationInsightSchema = z\n .object({\n leaks: nonNegativeInteger,\n holdoutAuditPassed: z.boolean(),\n details: z\n .array(\n z\n .object({\n runId: nonEmptyString,\n canary: nonEmptyString,\n matched: nonEmptyString,\n })\n .strict(),\n )\n .optional(),\n })\n .strict()\n\nconst OutcomeCorrelationInsightSchema = z\n .object({\n metric: nonEmptyString,\n n: nonNegativeInteger,\n pearson: finiteNumber,\n spearman: finiteNumber,\n rewardModel: z\n .object({\n intercept: finiteNumber,\n slope: finiteNumber,\n r2: finiteNumber,\n })\n .strict()\n .optional(),\n })\n .strict()\n\nconst ReleaseSummarySchema = z\n .object({\n status: z.enum(['pass', 'warn', 'fail']),\n axes: z.array(\n z\n .object({\n name: z.enum(['quality-lift', 'contamination', 'composite-distribution']),\n status: z.enum(['pass', 'warn', 'fail', 'not_evaluated']),\n detail: nonEmptyString,\n })\n .strict(),\n ),\n issues: z.array(z.string()),\n })\n .strict()\n\nconst MetricDeltaBaseSchema = z\n .object({\n current: finiteNumber,\n baseline: finiteNumber,\n delta: finiteNumber,\n baselineN: nonNegativeInteger,\n currentN: nonNegativeInteger,\n })\n .strict()\n\nconst MetricDeltaSchema = z.discriminatedUnion('status', [\n MetricDeltaBaseSchema.extend({\n status: z.literal('ok'),\n ci95: z.tuple([finiteNumber, finiteNumber]),\n pValue: finiteNumber.min(0).max(1),\n cohensD: finiteNumber,\n significant: z.boolean(),\n }),\n MetricDeltaBaseSchema.extend({\n status: z.enum(['insufficient-sample', 'zero-variance']),\n ci95: z.null(),\n pValue: z.null(),\n cohensD: z.null(),\n significant: z.literal(false),\n }),\n])\n\nconst PriorPeriodComparisonSchema = z\n .object({\n baselineN: nonNegativeInteger,\n currentN: nonNegativeInteger,\n windowLabel: nonEmptyString.optional(),\n metrics: z.record(z.string(), MetricDeltaSchema),\n regressedMetrics: z.array(nonEmptyString),\n improvedMetrics: z.array(nonEmptyString),\n inconclusiveMetrics: z.array(nonEmptyString),\n })\n .strict()\n\nconst RecommendationSchema = z\n .object({\n priority: z.enum(['critical', 'high', 'medium', 'low']),\n kind: z.enum(['ship', 'hold', 'investigate', 'fix', 'recalibrate', 'expand-corpus']),\n title: nonEmptyString,\n detail: nonEmptyString,\n evidencePath: nonEmptyString.optional(),\n })\n .strict()\n\nexport const InsightReportSchema: z.ZodType<InsightReport> = z\n .object({\n n: nonNegativeInteger,\n execution: ExecutionInsightSchema,\n composite: ScalarDistributionSchema,\n perDimension: z.record(z.string(), ScalarDistributionSchema),\n costQuality: z\n .object({\n cost: ScalarDistributionSchema,\n pareto: ParetoFigureSpecSchema,\n provenance: CostProvenanceSummarySchema.optional(),\n degraded: z\n .object({\n cost: nonEmptyString.optional(),\n pareto: nonEmptyString.optional(),\n })\n .strict()\n .optional(),\n })\n .strict(),\n judges: z.record(z.string(), JudgeInsightSchema),\n interRater: InterRaterInsightSchema.optional(),\n lift: LiftInsightSchema.optional(),\n failureClusters: FailureClusterInsightSchema.optional(),\n contamination: ContaminationInsightSchema.optional(),\n outcomeCorrelation: OutcomeCorrelationInsightSchema.optional(),\n release: ReleaseSummarySchema,\n priorPeriodComparison: PriorPeriodComparisonSchema.optional(),\n failureClasses: z\n .array(\n z\n .object({\n failureClass: FailureClassSchema,\n count: nonNegativeInteger,\n share: finiteNumber.min(0).max(1),\n })\n .strict(),\n )\n .optional(),\n recommendations: z.array(RecommendationSchema),\n })\n .strict()\n\nconst sha256Digest = z.custom<`sha256:${string}`>(\n (value) => typeof value === 'string' && /^sha256:[0-9a-f]{64}$/.test(value),\n 'expected sha256:<64 lowercase hex characters>',\n)\n\nexport const MutableSurfaceSchema: z.ZodType<MutableSurface> = z.union([\n z.string(),\n z\n .object({\n kind: z.literal('components'),\n components: z.record(z.string(), z.string()),\n })\n .strict(),\n z\n .object({\n kind: z.literal('code'),\n worktreeRef: nonEmptyString,\n baseRef: nonEmptyString,\n baseCommit: nonEmptyString,\n baseTree: nonEmptyString,\n candidateCommit: nonEmptyString,\n candidateTree: nonEmptyString,\n patch: z\n .object({\n format: z.literal('git-diff-binary'),\n sha256: sha256Digest,\n byteLength: nonNegativeInteger,\n })\n .strict(),\n summary: z.string().optional(),\n })\n .strict(),\n])\n\nexport const RunTerminalOutcomeSchema = z.enum([\n 'succeeded',\n 'failed',\n 'cancelled',\n 'incomplete',\n 'unknown',\n])\n\nexport const EvalRunCellScoreSchema: z.ZodType<EvalRunCellScore> = z\n .object({\n scenarioId: nonEmptyString,\n rep: nonNegativeInteger,\n compositeMean: finiteNumber.nullable(),\n dimensions: z.record(z.string(), z.record(z.string(), finiteNumber)),\n terminalOutcome: RunTerminalOutcomeSchema,\n executionErrorCount: nonNegativeInteger.nullable(),\n errorMessage: z.string().optional(),\n })\n .strict()\n\nexport const EvalRunGenerationSnapshotSchema: z.ZodType<EvalRunGenerationSnapshot> = z\n .object({\n index: nonNegativeInteger,\n surfaceHash: nonEmptyString,\n surface: MutableSurfaceSchema.optional(),\n cells: z.array(EvalRunCellScoreSchema),\n compositeMean: finiteNumber.nullable(),\n costUsd: nonNegativeNumber,\n durationMs: nonNegativeNumber,\n })\n .strict()\n\nconst EvalRunStatusSchema = z.enum([\n 'started',\n 'baseline-complete',\n 'generation-complete',\n 'gate-decided',\n 'finished',\n 'errored',\n])\n\nconst GateDecisionSchema = z.enum([\n 'ship',\n 'hold',\n 'need_more_work',\n 'model_ceiling',\n 'arch_ceiling',\n])\n\nexport const EvalRunEventSchema: z.ZodType<EvalRunEvent> = z\n .object({\n runId: nonEmptyString,\n runDir: nonEmptyString,\n timestamp: z.string().datetime({ offset: true }),\n status: EvalRunStatusSchema,\n labels: z.record(z.string(), z.string()),\n baseline: EvalRunGenerationSnapshotSchema.optional(),\n generations: z.array(EvalRunGenerationSnapshotSchema),\n gateDecision: GateDecisionSchema.optional(),\n holdoutLift: finiteNumber.optional(),\n totalCostUsd: nonNegativeNumber,\n totalDurationMs: nonNegativeNumber,\n errorMessage: z.string().optional(),\n insightReport: InsightReportSchema.optional(),\n })\n .strict()\n .superRefine((event, ctx) => {\n if (event.baseline && event.baseline.index !== 0) {\n ctx.addIssue({\n code: 'custom',\n path: ['baseline', 'index'],\n message: 'baseline index must be 0',\n })\n }\n\n const seen = new Set<number>()\n for (let i = 0; i < event.generations.length; i++) {\n const index = event.generations[i]!.index\n if (seen.has(index)) {\n ctx.addIssue({\n code: 'custom',\n path: ['generations', i, 'index'],\n message: `duplicate generation index ${index}`,\n })\n }\n seen.add(index)\n }\n\n if (event.status === 'errored' && !event.errorMessage?.trim()) {\n ctx.addIssue({\n code: 'custom',\n path: ['errorMessage'],\n message: 'errorMessage is required when status is errored',\n })\n }\n })\n\nconst TraceSpanEventEntrySchema = z\n .object({\n timeUnixNano: UnixNanoTimestampSchema,\n name: nonEmptyString,\n attributes: attributes.optional(),\n })\n .strict()\n\nexport const TraceSpanEventSchema: z.ZodType<TraceSpanEvent> = z\n .object({\n traceId: nonEmptyString,\n spanId: nonEmptyString,\n parentSpanId: nonEmptyString.optional(),\n name: nonEmptyString,\n startTimeUnixNano: UnixNanoTimestampSchema,\n endTimeUnixNano: UnixNanoTimestampSchema,\n attributes,\n events: z.array(TraceSpanEventEntrySchema).optional(),\n status: z\n .object({\n code: z.enum(['OK', 'ERROR', 'UNSET']),\n message: z.string().optional(),\n })\n .strict()\n .optional(),\n 'tangle.runId': nonEmptyString.optional(),\n 'tangle.generation': nonNegativeInteger.optional(),\n 'tangle.cellId': nonEmptyString.optional(),\n 'tangle.scenarioId': nonEmptyString.optional(),\n })\n .strict()\n .refine((span) => BigInt(span.endTimeUnixNano) >= BigInt(span.startTimeUnixNano), {\n path: ['endTimeUnixNano'],\n message: 'endTimeUnixNano must be greater than or equal to startTimeUnixNano',\n })\n\nexport const IngestEvalRunsEnvelopeSchema = z\n .object({\n wireVersion: z.literal(HOSTED_WIRE_VERSION),\n events: z.array(z.unknown()),\n })\n .strict()\n\nexport const IngestTracesEnvelopeSchema = z\n .object({\n wireVersion: z.literal(HOSTED_WIRE_VERSION),\n spans: z.array(z.unknown()),\n })\n .strict()\n\nexport const IngestEvalRunsRequestSchema: z.ZodType<IngestEvalRunsRequest> = z\n .object({\n wireVersion: z.literal(HOSTED_WIRE_VERSION),\n events: z.array(EvalRunEventSchema),\n })\n .strict()\n\nexport const IngestTracesRequestSchema: z.ZodType<IngestTracesRequest> = z\n .object({\n wireVersion: z.literal(HOSTED_WIRE_VERSION),\n spans: z.array(TraceSpanEventSchema),\n })\n .strict()\n\nexport const IngestResponseSchema: z.ZodType<IngestResponse> = z\n .object({\n accepted: nonNegativeInteger,\n rejected: z.array(\n z\n .object({\n index: nonNegativeInteger,\n reason: nonEmptyString,\n })\n .strict(),\n ),\n })\n .strict()\n","/**\n * # Hosted-tier ingest client.\n *\n * Ships eval-run events + trace spans to any orchestrator (ours, a\n * partner's self-hosted one, or a future open implementation) that\n * speaks the wire format in `./types.ts`.\n *\n * Three modes:\n * - **Ours:** point at `https://orchestrator.tangle.tools` (the host root —\n * the client appends the versioned `/v1/ingest/...` path itself; a trailing\n * `/v1` on the endpoint is tolerated and normalized away). We handle ingest\n * + storage + dashboard.\n * - **Self-hosted:** point at whatever URL runs the reference receiver\n * from `examples/hosted-ingest-server/`.\n * - **Off (default):** when `hostedTenant` is unset, nothing is sent.\n * Everything stays local.\n */\n\nimport {\n IngestEvalRunsRequestSchema,\n IngestResponseSchema,\n IngestTracesRequestSchema,\n} from './schemas'\nimport {\n type EvalRunEvent,\n HOSTED_WIRE_VERSION,\n type HostedWireVersion,\n type IngestEvalRunsRequest,\n type IngestResponse,\n type IngestTracesRequest,\n type TraceSpanEvent,\n} from './types'\n\nexport interface HostedTenant {\n /** Orchestrator endpoint base URL (no trailing slash). Required. */\n endpoint: string\n /** Bearer token issued by the orchestrator. Required. */\n apiKey: string\n /** Tenant id — the orchestrator's primary key for this consumer. Required. */\n tenantId: string\n /** Optional `fetch` override (auth wrappers, custom agent, test mocks). */\n fetchImpl?: typeof fetch\n /** Per-call timeout in ms. Default 30s. */\n timeoutMs?: number\n /** Retries on 5xx / network errors. Default 2. */\n retries?: number\n}\n\nexport interface HostedClient {\n ingestEvalRun(event: EvalRunEvent, idempotencyKey?: string): Promise<IngestResponse>\n ingestEvalRuns(events: EvalRunEvent[], idempotencyKey?: string): Promise<IngestResponse>\n ingestTraces(spans: TraceSpanEvent[], idempotencyKey?: string): Promise<IngestResponse>\n readonly tenant: HostedTenant\n readonly wireVersion: HostedWireVersion\n}\n\ninterface RequestOptions {\n idempotencyKey?: string\n signal?: AbortSignal\n}\n\nconst MAX_IDEMPOTENCY_KEY_LENGTH = 256\n\nfunction sleep(ms: number): Promise<void> {\n return new Promise((resolve) => setTimeout(resolve, ms))\n}\n\nfunction normalizeHostedBase(endpoint: string): string {\n return endpoint.trim().replace(/\\/+$/, '').replace(/\\/v1$/, '')\n}\n\nfunction resolveIdempotencyKey(key: string | undefined): string {\n const resolved = key ?? globalThis.crypto.randomUUID()\n if (resolved.trim().length === 0) throw new Error('idempotency key must not be blank')\n if (resolved.length > MAX_IDEMPOTENCY_KEY_LENGTH) {\n throw new Error(`idempotency key must be at most ${MAX_IDEMPOTENCY_KEY_LENGTH} characters`)\n }\n return resolved\n}\n\nfunction responseValidationReason(error: {\n issues: Array<{ path: PropertyKey[]; message: string }>\n}) {\n return error.issues\n .map(\n (issue) =>\n `${issue.path.length > 0 ? issue.path.map(String).join('.') : 'value'}: ${issue.message}`,\n )\n .join('; ')\n}\n\nasync function post<TReq>(\n tenant: HostedTenant,\n path: string,\n body: TReq,\n opts: RequestOptions = {},\n): Promise<IngestResponse> {\n const timeoutMs = tenant.timeoutMs ?? 30_000\n const maxRetries = tenant.retries ?? 2\n const f: typeof fetch = tenant.fetchImpl ?? ((...args) => fetch(...args))\n const base = normalizeHostedBase(tenant.endpoint)\n const url = `${base}${path}`\n const idempotencyKey = resolveIdempotencyKey(opts.idempotencyKey)\n const headers: Record<string, string> = {\n 'content-type': 'application/json',\n authorization: `Bearer ${tenant.apiKey}`,\n 'x-tangle-tenant-id': tenant.tenantId,\n 'x-tangle-wire-version': HOSTED_WIRE_VERSION,\n 'idempotency-key': idempotencyKey,\n }\n\n let lastError: unknown\n for (let attempt = 0; attempt <= maxRetries; attempt++) {\n const ourTimeout = AbortSignal.timeout(timeoutMs)\n const combinedSignal = opts.signal ? AbortSignal.any([opts.signal, ourTimeout]) : ourTimeout\n let res: Response\n try {\n res = await f(url, {\n method: 'POST',\n headers,\n body: JSON.stringify(body),\n signal: combinedSignal,\n })\n } catch (err) {\n if (opts.signal?.aborted) throw err\n lastError = err\n if (attempt === maxRetries) throw err\n await sleep(2 ** attempt * 200 + Math.random() * 200)\n continue\n }\n\n if (!res.ok) {\n const text = await res.text().catch(() => '')\n const error = new Error(`hosted ingest ${url} failed (${res.status}): ${text.slice(0, 500)}`)\n const retryable = res.status >= 500 || res.status === 408 || res.status === 429\n if (!retryable || attempt === maxRetries) throw error\n lastError = error\n await sleep(2 ** attempt * 200 + Math.random() * 200)\n continue\n }\n\n let rawResponse: unknown\n try {\n rawResponse = await res.json()\n } catch (error) {\n throw new Error(`hosted ingest ${url} returned invalid JSON`, { cause: error })\n }\n const parsed = IngestResponseSchema.safeParse(rawResponse)\n if (!parsed.success) {\n throw new Error(\n `hosted ingest ${url} returned an invalid response: ${responseValidationReason(parsed.error)}`,\n )\n }\n return parsed.data\n }\n throw lastError ?? new Error('hosted ingest exhausted retries')\n}\n\nexport function createHostedClient(tenant: HostedTenant): HostedClient {\n if (normalizeHostedBase(tenant.endpoint).length === 0) throw new Error('endpoint is required')\n if (tenant.apiKey.trim().length === 0) throw new Error('apiKey is required')\n if (tenant.tenantId.trim().length === 0) throw new Error('tenantId is required')\n if (\n tenant.timeoutMs !== undefined &&\n (!Number.isFinite(tenant.timeoutMs) || tenant.timeoutMs <= 0)\n ) {\n throw new Error('timeoutMs must be greater than 0')\n }\n if (tenant.retries !== undefined && (!Number.isInteger(tenant.retries) || tenant.retries < 0)) {\n throw new Error('retries must be a non-negative integer')\n }\n\n return {\n tenant,\n wireVersion: HOSTED_WIRE_VERSION,\n\n async ingestEvalRun(event, idempotencyKey) {\n return this.ingestEvalRuns([event], idempotencyKey)\n },\n\n async ingestEvalRuns(events, idempotencyKey) {\n const body: IngestEvalRunsRequest = IngestEvalRunsRequestSchema.parse({\n wireVersion: HOSTED_WIRE_VERSION,\n events,\n })\n return post<IngestEvalRunsRequest>(tenant, '/v1/ingest/eval-runs', body, {\n idempotencyKey,\n })\n },\n\n async ingestTraces(spans, idempotencyKey) {\n const body: IngestTracesRequest = IngestTracesRequestSchema.parse({\n wireVersion: HOSTED_WIRE_VERSION,\n spans,\n })\n return post<IngestTracesRequest>(tenant, '/v1/ingest/traces', body, {\n idempotencyKey,\n })\n },\n }\n}\n\n/**\n * Build a `HostedClient` from environment, or `undefined` when ingest is not\n * configured — the canonical, fail-soft wiring every product uses so eval-run +\n * trace provenance lands in the Intelligence dashboard with ONE call:\n *\n * const hosted = hostedClientFromEnv()\n * // ...run the loop...\n * await emitLoopProvenance({ ..., hostedClient: hosted }) // no-op if undefined\n *\n * Returns `undefined` (NOT an error) when any of endpoint / apiKey / tenantId is\n * missing — so a product wires the ship call unconditionally and it stays a\n * no-op until the env is set. Env precedence:\n * - endpoint: `TANGLE_INGEST_URL` → `TANGLE_ORCHESTRATOR_URL`\n * - apiKey: `TANGLE_INGEST_API_KEY` → `TANGLE_API_KEY`\n * - tenantId: `TANGLE_TENANT_ID`\n * A trailing slash on the endpoint is stripped. Pass `overrides` to supply any\n * field directly (e.g. a fixed `tenantId` per product) — overrides win over env.\n */\n/**\n * Build a {@link HostedTenant} config from env — the input `selfImprove`'s\n * `hostedTenant` and `emitLoopProvenance` take. Same env precedence + overrides\n * as {@link hostedClientFromEnv}; returns `undefined` (not an error) when any of\n * endpoint / apiKey / tenantId is missing, so a product wires\n * `hostedTenant: hostedTenantFromEnv({ tenantId: 'my-agent' })` unconditionally\n * and it stays off until the env is set.\n */\nexport function hostedTenantFromEnv(\n overrides: Partial<HostedTenant> & { env?: Record<string, string | undefined> } = {},\n): HostedTenant | undefined {\n const env = overrides.env ?? process.env\n const endpoint = (\n overrides.endpoint ??\n env.TANGLE_INGEST_URL ??\n env.TANGLE_ORCHESTRATOR_URL\n )?.trim()\n const apiKey = (overrides.apiKey ?? env.TANGLE_INGEST_API_KEY ?? env.TANGLE_API_KEY)?.trim()\n const tenantId = (overrides.tenantId ?? env.TANGLE_TENANT_ID)?.trim()\n if (!endpoint || !apiKey || !tenantId) return undefined\n const tenant: HostedTenant = { endpoint: endpoint.replace(/\\/+$/, ''), apiKey, tenantId }\n if (overrides.fetchImpl) tenant.fetchImpl = overrides.fetchImpl\n if (overrides.timeoutMs !== undefined) tenant.timeoutMs = overrides.timeoutMs\n if (overrides.retries !== undefined) tenant.retries = overrides.retries\n return tenant\n}\n\nexport function hostedClientFromEnv(\n overrides: Partial<HostedTenant> & { env?: Record<string, string | undefined> } = {},\n): HostedClient | undefined {\n const tenant = hostedTenantFromEnv(overrides)\n return tenant ? createHostedClient(tenant) : undefined\n}\n"],"mappings":";;;AAgCA,MAAa,sBAAsB;;;AChBnC,MAAM,eAAe,EAAE,OAAO,CAAC,CAAC,OAAO;AACvC,MAAM,oBAAoB,aAAa,YAAY;AACnD,MAAM,qBAAqB,EAAE,OAAO,CAAC,CAAC,IAAI,CAAC,CAAC,YAAY;AACxD,MAAM,iBAAiB,EACpB,OAAO,CAAC,CACR,IAAI,CAAC,CAAC,CACN,QAAQ,UAAU,MAAM,KAAK,CAAC,CAAC,SAAS,GAAG,mBAAmB;AACjE,MAAM,iBAAiB,EAAE,MAAM;CAAC,EAAE,OAAO;CAAG;CAAc,EAAE,QAAQ;AAAC,CAAC;AACtE,MAAM,aAAa,EAAE,OAAO,EAAE,OAAO,GAAG,cAAc;AACtD,MAAM,qBAAqB,EAAE,QAC1B,UAAU,OAAO,UAAU,YAAY,gBAAgB,SAAS,KAAqB,GACtF,mBAAmB,gBAAgB,KAAK,IAAI,GAC9C;AACA,MAAM,aAAa;AAEnB,MAAa,0BAAwD,EAClE,OAAO,CAAC,CACR,MAAM,qBAAqB,6CAA6C,CAAC,CACzE,QAAQ,UAAU,OAAO,KAAK,KAAK,YAAY,wCAAwC;AAE1F,MAAM,4BAA4B,EAC/B,OAAO;CACN,IAAI;CACJ,IAAI;CACJ,OAAO;AACT,CAAC,CAAC,CACD,OAAO;AAEV,MAAM,2BAA2B,EAC9B,OAAO;CACN,GAAG;CACH,MAAM,aAAa,SAAS;CAC5B,KAAK,aAAa,SAAS;CAC3B,KAAK,aAAa,SAAS;CAC3B,QAAQ,kBAAkB,SAAS;CACnC,KAAK,aAAa,SAAS;CAC3B,KAAK,aAAa,SAAS;CAC3B,WAAW,EAAE,MAAM,yBAAyB;CAC5C,UAAU,EACP,MACC,EACG,OAAO;EACN,OAAO;EACP,OAAO;CACT,CAAC,CAAC,CACD,OAAO,CACZ,CAAC,CACA,SAAS;AACd,CAAC,CAAC,CACD,OAAO,CAAC,CACR,aAAa,cAAc,QAAQ;CAClC,MAAM,SAAS;EACb,aAAa;EACb,aAAa;EACb,aAAa;EACb,aAAa;EACb,aAAa;EACb,aAAa;CACf;CACA,IAAI,aAAa,MAAM,KAAK,OAAO,MAAM,UAAU,UAAU,IAAI,GAC/D,IAAI,SAAS;EACX,MAAM;EACN,SAAS;CACX,CAAC;CAEH,IAAI,aAAa,IAAI,KAAK,OAAO,MAAM,UAAU,UAAU,IAAI,GAC7D,IAAI,SAAS;EACX,MAAM;EACN,SAAS;CACX,CAAC;CAEH,IACE,aAAa,QAAQ,QACrB,aAAa,QAAQ,QACrB,aAAa,MAAM,aAAa,KAEhC,IAAI,SAAS;EAAE,MAAM;EAAU,MAAM,CAAC,KAAK;EAAG,SAAS;CAA0B,CAAC;AAEtF,CAAC;AAEH,MAAM,0BAA0B,EAC7B,OAAO;CACN,OAAO;CACP,QAAQ;CACR,WAAW;CACX,QAAQ;CACR,YAAY;CACZ,QAAQ,EACL,OAAO;EACN,OAAO;EACP,QAAQ;EACR,WAAW;EACX,QAAQ;EACR,YAAY;CACd,CAAC,CAAC,CACD,OAAO;AACZ,CAAC,CAAC,CACD,OAAO;AAEV,MAAM,kCAAkC,EACrC,OAAO;CACN,YAAY;CACZ,eAAe;CACf,YAAY;AACd,CAAC,CAAC,CACD,OAAO;AAEV,MAAM,yBAAyB,EAC5B,OAAO;CACN,YAAY;CACZ,SAAS;CACT,YAAY;CACZ,gBAAgB,EACb,OAAO;EACN,MAAM;EACN,YAAY;EACZ,SAAS;EACT,cAAc;CAChB,CAAC,CAAC,CACD,OAAO;CACV,QAAQ,EAAE,MACR,EACG,OAAO;EACN,OAAO;EACP,MAAM;CACR,CAAC,CAAC,CACD,OAAO,CACZ;CACA,YAAY,EACT,OAAO;EACN,MAAM;EACN,QAAQ;EACR,eAAe;CACjB,CAAC,CAAC,CACD,OAAO;CACV,iBAAiB,EACd,OAAO;EACN,MAAM;EACN,UAAU,aAAa,IAAI,CAAC,CAAC,CAAC,IAAI,CAAC,CAAC,CAAC,SAAS;EAC9C,QAAQ;EACR,eAAe;EACf,iBAAiB;EACjB,wBAAwB;EACxB,mBAAmB,EAChB,OAAO;GACN,WAAW;GACX,QAAQ;GACR,WAAW;GACX,YAAY;GACZ,SAAS;EACX,CAAC,CAAC,CACD,OAAO;CACZ,CAAC,CAAC,CACD,OAAO;CACV,kBAAkB,EACf,OAAO;EACN,WAAW;EACX,QAAQ;EACR,WAAW;EACX,YAAY;EACZ,SAAS;CACX,CAAC,CAAC,CACD,OAAO;AACZ,CAAC,CAAC,CACD,OAAO;AAEV,MAAM,8BAA8B,EACjC,OAAO;CACN,UAAU,EACP,OAAO;EACN,GAAG;EACH,UAAU;CACZ,CAAC,CAAC,CACD,OAAO;CACV,WAAW,EACR,OAAO;EACN,GAAG;EACH,UAAU;CACZ,CAAC,CAAC,CACD,OAAO;CACV,YAAY,EAAE,OAAO,EAAE,GAAG,mBAAmB,CAAC,CAAC,CAAC,OAAO;CACvD,eAAe,aAAa,IAAI,CAAC,CAAC,CAAC,IAAI,CAAC;AAC1C,CAAC,CAAC,CACD,OAAO;AAEV,MAAM,yBAAyB,EAC5B,OAAO;CACN,MAAM,EAAE,QAAQ,qBAAqB;CACrC,OAAO,EAAE,KAAK,CAAC,UAAU,SAAS,CAAC;CACnC,QAAQ,EAAE,MACR,EACG,OAAO;EACN,aAAa;EACb,MAAM;EACN,SAAS;EACT,GAAG;EACH,YAAY,EAAE,QAAQ;EACtB,MAAM,EAAE,KAAK,CAAC,WAAW,QAAQ,CAAC,CAAC,CAAC,SAAS;CAC/C,CAAC,CAAC,CACD,OAAO,CACZ;CACA,MAAM,EAAE,OAAO;EAAE,GAAG,EAAE,QAAQ,SAAS;EAAG,GAAG,EAAE,QAAQ,OAAO;CAAE,CAAC,CAAC,CAAC,OAAO;AAC5E,CAAC,CAAC,CACD,OAAO;AAEV,MAAM,4BAA4B,EAC/B,OAAO;CACN,eAAe;CACf,KAAK;CACL,SAAS;CACT,UAAU;CACV,IAAI,EACD,OAAO;EACN,KAAK,EAAE,MAAM,CAAC,cAAc,YAAY,CAAC;EACzC,eAAe,EAAE,MAAM,CAAC,cAAc,YAAY,CAAC;CACrD,CAAC,CAAC,CACD,OAAO;CACV,GAAG;CACH,QAAQ;AACV,CAAC,CAAC,CACD,OAAO;AAEV,MAAM,qBAAqB,EACxB,OAAO;CACN,GAAG;CACH,WAAW;CACX,aAAa,0BAA0B,SAAS;CAChD,gBAAgB,aAAa,SAAS;CACtC,gBAAgB,aAAa,SAAS;CACtC,eAAe,aAAa,SAAS;AACvC,CAAC,CAAC,CACD,OAAO;AAEV,MAAM,0BAA0B,EAC7B,OAAO;CACN,QAAQ;CACR,cAAc;CACd,OAAO;CACP,KAAK;CACL,SAAS;CACT,UAAU;CACV,SAAS,EAAE,OAAO,EAAE,OAAO,GAAG,YAAY;CAC1C,mBAAmB,EAAE,MACnB,EACG,OAAO;EACN,OAAO;EACP,SAAS,EAAE,MACT,EACG,OAAO;GACN,OAAO;GACP,OAAO;EACT,CAAC,CAAC,CACD,OAAO,CACZ;EACA,OAAO;CACT,CAAC,CAAC,CACD,OAAO,CACZ;AACF,CAAC,CAAC,CACD,OAAO;AAEV,MAAM,oBAAoB,EACvB,OAAO;CACN,cAAc;CACd,eAAe;CACf,OAAO;CACP,MAAM,EAAE,MAAM,CAAC,cAAc,YAAY,CAAC;CAC1C,QAAQ,aAAa,IAAI,CAAC,CAAC,CAAC,IAAI,CAAC,CAAC,CAAC,SAAS;CAC5C,GAAG;CACH,iBAAiB,EAAE,OAAO,CAAC,CAAC,IAAI,CAAC,CAAC,SAAS;CAC3C,kBAAkB,EAAE,QAAQ;CAC5B,kBAAkB;CAClB,mBAAmB;CACnB,SAAS,aAAa,SAAS;CAC/B,KAAK;CACL,WAAW,mBAAmB,SAAS;AACzC,CAAC,CAAC,CACD,OAAO;AAEV,MAAM,8BAA8B,EACjC,OAAO;CACN,UAAU,EAAE,MACV,EACG,OAAO;EACN,IAAI;EACJ,MAAM;EACN,OAAO,aAAa,IAAI,CAAC,CAAC,CAAC,IAAI,CAAC;EAChC,WAAW,EAAE,MAAM,cAAc,CAAC,CAAC,IAAI,CAAC;EACxC,cAAc,eAAe,SAAS;CACxC,CAAC,CAAC,CACD,OAAO,CACZ;CACA,eAAe;AACjB,CAAC,CAAC,CACD,OAAO;AAEV,MAAM,6BAA6B,EAChC,OAAO;CACN,OAAO;CACP,oBAAoB,EAAE,QAAQ;CAC9B,SAAS,EACN,MACC,EACG,OAAO;EACN,OAAO;EACP,QAAQ;EACR,SAAS;CACX,CAAC,CAAC,CACD,OAAO,CACZ,CAAC,CACA,SAAS;AACd,CAAC,CAAC,CACD,OAAO;AAEV,MAAM,kCAAkC,EACrC,OAAO;CACN,QAAQ;CACR,GAAG;CACH,SAAS;CACT,UAAU;CACV,aAAa,EACV,OAAO;EACN,WAAW;EACX,OAAO;EACP,IAAI;CACN,CAAC,CAAC,CACD,OAAO,CAAC,CACR,SAAS;AACd,CAAC,CAAC,CACD,OAAO;AAEV,MAAM,uBAAuB,EAC1B,OAAO;CACN,QAAQ,EAAE,KAAK;EAAC;EAAQ;EAAQ;CAAM,CAAC;CACvC,MAAM,EAAE,MACN,EACG,OAAO;EACN,MAAM,EAAE,KAAK;GAAC;GAAgB;GAAiB;EAAwB,CAAC;EACxE,QAAQ,EAAE,KAAK;GAAC;GAAQ;GAAQ;GAAQ;EAAe,CAAC;EACxD,QAAQ;CACV,CAAC,CAAC,CACD,OAAO,CACZ;CACA,QAAQ,EAAE,MAAM,EAAE,OAAO,CAAC;AAC5B,CAAC,CAAC,CACD,OAAO;AAEV,MAAM,wBAAwB,EAC3B,OAAO;CACN,SAAS;CACT,UAAU;CACV,OAAO;CACP,WAAW;CACX,UAAU;AACZ,CAAC,CAAC,CACD,OAAO;AAEV,MAAM,oBAAoB,EAAE,mBAAmB,UAAU,CACvD,sBAAsB,OAAO;CAC3B,QAAQ,EAAE,QAAQ,IAAI;CACtB,MAAM,EAAE,MAAM,CAAC,cAAc,YAAY,CAAC;CAC1C,QAAQ,aAAa,IAAI,CAAC,CAAC,CAAC,IAAI,CAAC;CACjC,SAAS;CACT,aAAa,EAAE,QAAQ;AACzB,CAAC,GACD,sBAAsB,OAAO;CAC3B,QAAQ,EAAE,KAAK,CAAC,uBAAuB,eAAe,CAAC;CACvD,MAAM,EAAE,KAAK;CACb,QAAQ,EAAE,KAAK;CACf,SAAS,EAAE,KAAK;CAChB,aAAa,EAAE,QAAQ,KAAK;AAC9B,CAAC,CACH,CAAC;AAED,MAAM,8BAA8B,EACjC,OAAO;CACN,WAAW;CACX,UAAU;CACV,aAAa,eAAe,SAAS;CACrC,SAAS,EAAE,OAAO,EAAE,OAAO,GAAG,iBAAiB;CAC/C,kBAAkB,EAAE,MAAM,cAAc;CACxC,iBAAiB,EAAE,MAAM,cAAc;CACvC,qBAAqB,EAAE,MAAM,cAAc;AAC7C,CAAC,CAAC,CACD,OAAO;AAEV,MAAM,uBAAuB,EAC1B,OAAO;CACN,UAAU,EAAE,KAAK;EAAC;EAAY;EAAQ;EAAU;CAAK,CAAC;CACtD,MAAM,EAAE,KAAK;EAAC;EAAQ;EAAQ;EAAe;EAAO;EAAe;CAAe,CAAC;CACnF,OAAO;CACP,QAAQ;CACR,cAAc,eAAe,SAAS;AACxC,CAAC,CAAC,CACD,OAAO;AAEV,MAAa,sBAAgD,EAC1D,OAAO;CACN,GAAG;CACH,WAAW;CACX,WAAW;CACX,cAAc,EAAE,OAAO,EAAE,OAAO,GAAG,wBAAwB;CAC3D,aAAa,EACV,OAAO;EACN,MAAM;EACN,QAAQ;EACR,YAAY,4BAA4B,SAAS;EACjD,UAAU,EACP,OAAO;GACN,MAAM,eAAe,SAAS;GAC9B,QAAQ,eAAe,SAAS;EAClC,CAAC,CAAC,CACD,OAAO,CAAC,CACR,SAAS;CACd,CAAC,CAAC,CACD,OAAO;CACV,QAAQ,EAAE,OAAO,EAAE,OAAO,GAAG,kBAAkB;CAC/C,YAAY,wBAAwB,SAAS;CAC7C,MAAM,kBAAkB,SAAS;CACjC,iBAAiB,4BAA4B,SAAS;CACtD,eAAe,2BAA2B,SAAS;CACnD,oBAAoB,gCAAgC,SAAS;CAC7D,SAAS;CACT,uBAAuB,4BAA4B,SAAS;CAC5D,gBAAgB,EACb,MACC,EACG,OAAO;EACN,cAAc;EACd,OAAO;EACP,OAAO,aAAa,IAAI,CAAC,CAAC,CAAC,IAAI,CAAC;CAClC,CAAC,CAAC,CACD,OAAO,CACZ,CAAC,CACA,SAAS;CACZ,iBAAiB,EAAE,MAAM,oBAAoB;AAC/C,CAAC,CAAC,CACD,OAAO;AAEV,MAAM,eAAe,EAAE,QACpB,UAAU,OAAO,UAAU,YAAY,wBAAwB,KAAK,KAAK,GAC1E,+CACF;AAEA,MAAa,uBAAkD,EAAE,MAAM;CACrE,EAAE,OAAO;CACT,EACG,OAAO;EACN,MAAM,EAAE,QAAQ,YAAY;EAC5B,YAAY,EAAE,OAAO,EAAE,OAAO,GAAG,EAAE,OAAO,CAAC;CAC7C,CAAC,CAAC,CACD,OAAO;CACV,EACG,OAAO;EACN,MAAM,EAAE,QAAQ,MAAM;EACtB,aAAa;EACb,SAAS;EACT,YAAY;EACZ,UAAU;EACV,iBAAiB;EACjB,eAAe;EACf,OAAO,EACJ,OAAO;GACN,QAAQ,EAAE,QAAQ,iBAAiB;GACnC,QAAQ;GACR,YAAY;EACd,CAAC,CAAC,CACD,OAAO;EACV,SAAS,EAAE,OAAO,CAAC,CAAC,SAAS;CAC/B,CAAC,CAAC,CACD,OAAO;AACZ,CAAC;AAED,MAAa,2BAA2B,EAAE,KAAK;CAC7C;CACA;CACA;CACA;CACA;AACF,CAAC;AAED,MAAa,yBAAsD,EAChE,OAAO;CACN,YAAY;CACZ,KAAK;CACL,eAAe,aAAa,SAAS;CACrC,YAAY,EAAE,OAAO,EAAE,OAAO,GAAG,EAAE,OAAO,EAAE,OAAO,GAAG,YAAY,CAAC;CACnE,iBAAiB;CACjB,qBAAqB,mBAAmB,SAAS;CACjD,cAAc,EAAE,OAAO,CAAC,CAAC,SAAS;AACpC,CAAC,CAAC,CACD,OAAO;AAEV,MAAa,kCAAwE,EAClF,OAAO;CACN,OAAO;CACP,aAAa;CACb,SAAS,qBAAqB,SAAS;CACvC,OAAO,EAAE,MAAM,sBAAsB;CACrC,eAAe,aAAa,SAAS;CACrC,SAAS;CACT,YAAY;AACd,CAAC,CAAC,CACD,OAAO;AAEV,MAAM,sBAAsB,EAAE,KAAK;CACjC;CACA;CACA;CACA;CACA;CACA;AACF,CAAC;AAED,MAAM,qBAAqB,EAAE,KAAK;CAChC;CACA;CACA;CACA;CACA;AACF,CAAC;AAED,MAAa,qBAA8C,EACxD,OAAO;CACN,OAAO;CACP,QAAQ;CACR,WAAW,EAAE,OAAO,CAAC,CAAC,SAAS,EAAE,QAAQ,KAAK,CAAC;CAC/C,QAAQ;CACR,QAAQ,EAAE,OAAO,EAAE,OAAO,GAAG,EAAE,OAAO,CAAC;CACvC,UAAU,gCAAgC,SAAS;CACnD,aAAa,EAAE,MAAM,+BAA+B;CACpD,cAAc,mBAAmB,SAAS;CAC1C,aAAa,aAAa,SAAS;CACnC,cAAc;CACd,iBAAiB;CACjB,cAAc,EAAE,OAAO,CAAC,CAAC,SAAS;CAClC,eAAe,oBAAoB,SAAS;AAC9C,CAAC,CAAC,CACD,OAAO,CAAC,CACR,aAAa,OAAO,QAAQ;CAC3B,IAAI,MAAM,YAAY,MAAM,SAAS,UAAU,GAC7C,IAAI,SAAS;EACX,MAAM;EACN,MAAM,CAAC,YAAY,OAAO;EAC1B,SAAS;CACX,CAAC;CAGH,MAAM,uBAAO,IAAI,IAAY;CAC7B,KAAK,IAAI,IAAI,GAAG,IAAI,MAAM,YAAY,QAAQ,KAAK;EACjD,MAAM,QAAQ,MAAM,YAAY,EAAE,CAAE;EACpC,IAAI,KAAK,IAAI,KAAK,GAChB,IAAI,SAAS;GACX,MAAM;GACN,MAAM;IAAC;IAAe;IAAG;GAAO;GAChC,SAAS,8BAA8B;EACzC,CAAC;EAEH,KAAK,IAAI,KAAK;CAChB;CAEA,IAAI,MAAM,WAAW,aAAa,CAAC,MAAM,cAAc,KAAK,GAC1D,IAAI,SAAS;EACX,MAAM;EACN,MAAM,CAAC,cAAc;EACrB,SAAS;CACX,CAAC;AAEL,CAAC;AAEH,MAAM,4BAA4B,EAC/B,OAAO;CACN,cAAc;CACd,MAAM;CACN,YAAY,WAAW,SAAS;AAClC,CAAC,CAAC,CACD,OAAO;AAEV,MAAa,uBAAkD,EAC5D,OAAO;CACN,SAAS;CACT,QAAQ;CACR,cAAc,eAAe,SAAS;CACtC,MAAM;CACN,mBAAmB;CACnB,iBAAiB;CACjB;CACA,QAAQ,EAAE,MAAM,yBAAyB,CAAC,CAAC,SAAS;CACpD,QAAQ,EACL,OAAO;EACN,MAAM,EAAE,KAAK;GAAC;GAAM;GAAS;EAAO,CAAC;EACrC,SAAS,EAAE,OAAO,CAAC,CAAC,SAAS;CAC/B,CAAC,CAAC,CACD,OAAO,CAAC,CACR,SAAS;CACZ,gBAAgB,eAAe,SAAS;CACxC,qBAAqB,mBAAmB,SAAS;CACjD,iBAAiB,eAAe,SAAS;CACzC,qBAAqB,eAAe,SAAS;AAC/C,CAAC,CAAC,CACD,OAAO,CAAC,CACR,QAAQ,SAAS,OAAO,KAAK,eAAe,KAAK,OAAO,KAAK,iBAAiB,GAAG;CAChF,MAAM,CAAC,iBAAiB;CACxB,SAAS;AACX,CAAC;AAEyC,EACzC,OAAO;CACN,aAAa,EAAE,QAAQ,mBAAmB;CAC1C,QAAQ,EAAE,MAAM,EAAE,QAAQ,CAAC;AAC7B,CAAC,CAAC,CACD,OAAO;AAEgC,EACvC,OAAO;CACN,aAAa,EAAE,QAAQ,mBAAmB;CAC1C,OAAO,EAAE,MAAM,EAAE,QAAQ,CAAC;AAC5B,CAAC,CAAC,CACD,OAAO;AAEV,MAAa,8BAAgE,EAC1E,OAAO;CACN,aAAa,EAAE,QAAQ,mBAAmB;CAC1C,QAAQ,EAAE,MAAM,kBAAkB;AACpC,CAAC,CAAC,CACD,OAAO;AAEV,MAAa,4BAA4D,EACtE,OAAO;CACN,aAAa,EAAE,QAAQ,mBAAmB;CAC1C,OAAO,EAAE,MAAM,oBAAoB;AACrC,CAAC,CAAC,CACD,OAAO;AAEV,MAAa,uBAAkD,EAC5D,OAAO;CACN,UAAU;CACV,UAAU,EAAE,MACV,EACG,OAAO;EACN,OAAO;EACP,QAAQ;CACV,CAAC,CAAC,CACD,OAAO,CACZ;AACF,CAAC,CAAC,CACD,OAAO;;;;;;;;;;;;;;;;;;;;ACzlBV,MAAM,6BAA6B;AAEnC,SAAS,MAAM,IAA2B;CACxC,OAAO,IAAI,SAAS,YAAY,WAAW,SAAS,EAAE,CAAC;AACzD;AAEA,SAAS,oBAAoB,UAA0B;CACrD,OAAO,SAAS,KAAK,CAAC,CAAC,QAAQ,QAAQ,EAAE,CAAC,CAAC,QAAQ,SAAS,EAAE;AAChE;AAEA,SAAS,sBAAsB,KAAiC;CAC9D,MAAM,WAAW,OAAO,WAAW,OAAO,WAAW;CACrD,IAAI,SAAS,KAAK,CAAC,CAAC,WAAW,GAAG,MAAM,IAAI,MAAM,mCAAmC;CACrF,IAAI,SAAS,SAAS,4BACpB,MAAM,IAAI,MAAM,mCAAmC,2BAA2B,YAAY;CAE5F,OAAO;AACT;AAEA,SAAS,yBAAyB,OAE/B;CACD,OAAO,MAAM,OACV,KACE,UACC,GAAG,MAAM,KAAK,SAAS,IAAI,MAAM,KAAK,IAAI,MAAM,CAAC,CAAC,KAAK,GAAG,IAAI,QAAQ,IAAI,MAAM,SACpF,CAAC,CACA,KAAK,IAAI;AACd;AAEA,eAAe,KACb,QACA,MACA,MACA,OAAuB,CAAC,GACC;CACzB,MAAM,YAAY,OAAO,aAAa;CACtC,MAAM,aAAa,OAAO,WAAW;CACrC,MAAM,IAAkB,OAAO,eAAe,GAAG,SAAS,MAAM,GAAG,IAAI;CAEvE,MAAM,MAAM,GADC,oBAAoB,OAAO,QACtB,IAAI;CACtB,MAAM,iBAAiB,sBAAsB,KAAK,cAAc;CAChE,MAAM,UAAkC;EACtC,gBAAgB;EAChB,eAAe,UAAU,OAAO;EAChC,sBAAsB,OAAO;EAC7B,yBAAyB;EACzB,mBAAmB;CACrB;CAEA,IAAI;CACJ,KAAK,IAAI,UAAU,GAAG,WAAW,YAAY,WAAW;EACtD,MAAM,aAAa,YAAY,QAAQ,SAAS;EAChD,MAAM,iBAAiB,KAAK,SAAS,YAAY,IAAI,CAAC,KAAK,QAAQ,UAAU,CAAC,IAAI;EAClF,IAAI;EACJ,IAAI;GACF,MAAM,MAAM,EAAE,KAAK;IACjB,QAAQ;IACR;IACA,MAAM,KAAK,UAAU,IAAI;IACzB,QAAQ;GACV,CAAC;EACH,SAAS,KAAK;GACZ,IAAI,KAAK,QAAQ,SAAS,MAAM;GAChC,YAAY;GACZ,IAAI,YAAY,YAAY,MAAM;GAClC,MAAM,MAAM,KAAK,UAAU,MAAM,KAAK,OAAO,IAAI,GAAG;GACpD;EACF;EAEA,IAAI,CAAC,IAAI,IAAI;GACX,MAAM,OAAO,MAAM,IAAI,KAAK,CAAC,CAAC,YAAY,EAAE;GAC5C,MAAM,wBAAQ,IAAI,MAAM,iBAAiB,IAAI,WAAW,IAAI,OAAO,KAAK,KAAK,MAAM,GAAG,GAAG,GAAG;GAE5F,IAAI,EADc,IAAI,UAAU,OAAO,IAAI,WAAW,OAAO,IAAI,WAAW,QAC1D,YAAY,YAAY,MAAM;GAChD,YAAY;GACZ,MAAM,MAAM,KAAK,UAAU,MAAM,KAAK,OAAO,IAAI,GAAG;GACpD;EACF;EAEA,IAAI;EACJ,IAAI;GACF,cAAc,MAAM,IAAI,KAAK;EAC/B,SAAS,OAAO;GACd,MAAM,IAAI,MAAM,iBAAiB,IAAI,yBAAyB,EAAE,OAAO,MAAM,CAAC;EAChF;EACA,MAAM,SAAS,qBAAqB,UAAU,WAAW;EACzD,IAAI,CAAC,OAAO,SACV,MAAM,IAAI,MACR,iBAAiB,IAAI,iCAAiC,yBAAyB,OAAO,KAAK,GAC7F;EAEF,OAAO,OAAO;CAChB;CACA,MAAM,6BAAa,IAAI,MAAM,iCAAiC;AAChE;AAEA,SAAgB,mBAAmB,QAAoC;CACrE,IAAI,oBAAoB,OAAO,QAAQ,CAAC,CAAC,WAAW,GAAG,MAAM,IAAI,MAAM,sBAAsB;CAC7F,IAAI,OAAO,OAAO,KAAK,CAAC,CAAC,WAAW,GAAG,MAAM,IAAI,MAAM,oBAAoB;CAC3E,IAAI,OAAO,SAAS,KAAK,CAAC,CAAC,WAAW,GAAG,MAAM,IAAI,MAAM,sBAAsB;CAC/E,IACE,OAAO,cAAc,KAAA,MACpB,CAAC,OAAO,SAAS,OAAO,SAAS,KAAK,OAAO,aAAa,IAE3D,MAAM,IAAI,MAAM,kCAAkC;CAEpD,IAAI,OAAO,YAAY,KAAA,MAAc,CAAC,OAAO,UAAU,OAAO,OAAO,KAAK,OAAO,UAAU,IACzF,MAAM,IAAI,MAAM,wCAAwC;CAG1D,OAAO;EACL;EACA,aAAa;EAEb,MAAM,cAAc,OAAO,gBAAgB;GACzC,OAAO,KAAK,eAAe,CAAC,KAAK,GAAG,cAAc;EACpD;EAEA,MAAM,eAAe,QAAQ,gBAAgB;GAK3C,OAAO,KAA4B,QAAQ,wBAJP,4BAA4B,MAAM;IACpE,aAAa;IACb;GACF,CACsE,GAAG,EACvE,eACF,CAAC;EACH;EAEA,MAAM,aAAa,OAAO,gBAAgB;GAKxC,OAAO,KAA0B,QAAQ,qBAJP,0BAA0B,MAAM;IAChE,aAAa;IACb;GACF,CACiE,GAAG,EAClE,eACF,CAAC;EACH;CACF;AACF;;;;;;;;;;;;;;;;;;;;;;;;;;;AA4BA,SAAgB,oBACd,YAAkF,CAAC,GACzD;CAC1B,MAAM,MAAM,UAAU,OAAO,QAAQ;CACrC,MAAM,YACJ,UAAU,YACV,IAAI,qBACJ,IAAI,wBAAA,EACH,KAAK;CACR,MAAM,UAAU,UAAU,UAAU,IAAI,yBAAyB,IAAI,eAAA,EAAiB,KAAK;CAC3F,MAAM,YAAY,UAAU,YAAY,IAAI,iBAAA,EAAmB,KAAK;CACpE,IAAI,CAAC,YAAY,CAAC,UAAU,CAAC,UAAU,OAAO,KAAA;CAC9C,MAAM,SAAuB;EAAE,UAAU,SAAS,QAAQ,QAAQ,EAAE;EAAG;EAAQ;CAAS;CACxF,IAAI,UAAU,WAAW,OAAO,YAAY,UAAU;CACtD,IAAI,UAAU,cAAc,KAAA,GAAW,OAAO,YAAY,UAAU;CACpE,IAAI,UAAU,YAAY,KAAA,GAAW,OAAO,UAAU,UAAU;CAChE,OAAO;AACT;AAEA,SAAgB,oBACd,YAAkF,CAAC,GACzD;CAC1B,MAAM,SAAS,oBAAoB,SAAS;CAC5C,OAAO,SAAS,mBAAmB,MAAM,IAAI,KAAA;AAC/C"}
|
package/dist/contract/index.d.ts
CHANGED
|
@@ -1,12 +1,13 @@
|
|
|
1
1
|
import { c as CostLedgerHandle, f as CostLedgerSummary, m as CostReceipt } from "../cost-ledger-fGS_u_O1.js";
|
|
2
2
|
import { a as RunRecord, n as RunCostProvenance, s as RunSplitTag } from "../run-record-DcObtIGh.js";
|
|
3
3
|
import { A as ChatClient, F as CreateChatClientOpts, V as createChatClient } from "../types-Cc3qbqzj.js";
|
|
4
|
-
import {
|
|
5
|
-
import {
|
|
6
|
-
import {
|
|
7
|
-
import { A as
|
|
4
|
+
import { h as ProposalFindingOrigin, i as AnalystFinding, m as ProposalFinding, v as makeProposalFinding } from "../types-DVjczBM9.js";
|
|
5
|
+
import { n as buildDefaultAnalystRegistry, t as DefaultAnalystRegistryOptions } from "../default-registry-Brxr728w.js";
|
|
6
|
+
import { C as JudgeDimension, H as SurfaceProposer, M as OptimizerConfig, R as Scenario, S as JudgeConfig, V as SessionScript, _ as GateDecision, a as CampaignResult, b as GenerationRecord, c as CampaignTraceWriter, d as DispatchContext, f as DispatchFn, g as GateContribution, h as GateContext, i as CampaignCostMeter, j as MutableSurface, k as LabeledScenarioStore, l as CodeSurface, m as GateCheckStatus, n as CampaignArtifactWriter, p as Gate, r as CampaignCellResult, t as CampaignAggregates, v as GateResult, w as JudgeScore, y as GenerationCandidate } from "../types-DiWLru6Z.js";
|
|
7
|
+
import { A as RunEvalOptions, B as OptimizerModelBudget, Bt as ComparisonCost, C as RunImprovementLoopOptions, Cn as ReferenceEquivalenceJudgeResult, D as RunOptimizationOptions, Dn as LlmJudgeDimension, E as PremeasuredOptimizationBaseline, En as runReferenceEquivalenceJudge, F as GepaOptimizationMethodConfig, Ft as ExternalTextOptimizerContext, G as ObjectiveSource, Gt as OptimizationMethodProvenance, H as AxisVerdict, Ht as OptimizationMethodComparison, I as GepaOptimizationRecipe, It as ExternalTextOptimizerResult, J as PromotionPolicy, K as ParetoSignificanceGateOptions, Kt as OptimizationMethodResult, L as GepaRunnerCommand, Lt as ExternalOptimizationExample, M as GepaAdaptiveEngineRun, Mt as composeGate, N as GepaEngineOptions, Nt as externalTextOptimizationMethod, On as LlmJudgeOptions, P as GepaEngineRun, Pt as ExternalTextOptimizationMethodConfig, R as gepaOptimizationMethod, Rt as ExternalTextEvaluationResponse, Sn as ReferenceEquivalenceJudgeOptions, T as runImprovementLoop, Tn as createReferenceEquivalenceJudge, U as BuildEvidenceVectorOptions, Ut as OptimizationMethodInput, V as AxisEvidence, Vt as OptimizationMethod, W as EvidenceVector, X as paretoPolicy, Xt as OptimizationTokenUsage, Y as buildEvidenceVector, Yt as OptimizationPackageSource, Z as paretoSignificanceGate, Zt as compareOptimizationMethods, bn as REFERENCE_EQUIVALENCE_JUDGE_VERSION, dt as DefaultProductionGateCheck, en as CampaignCellFailureReceipt, ft as DefaultProductionGateOptions, i as skillOptOptimizationMethod, in as RunCampaignOptions, j as runEval, kn as llmJudge, ln as fsCampaignStorage, lt as HeldOutGateOptions, mn as campaignSplitDigest, mt as defaultProductionGate, n as SkillOptRunnerCommand, on as runCampaign, ot as PowerPreflight, p as LoopProvenanceRecord, pt as DefaultProductionRewardHackingOptions, q as PromotionObjective, r as SkillOptTrainerConfig, sn as CampaignStorage, t as SkillOptOptimizationMethodConfig, un as inMemoryCampaignStorage, ut as heldOutGate, w as RunImprovementLoopResult, wn as ReferenceEquivalenceScenario, xn as ReferenceEquivalenceJudgeInput, yn as REFERENCE_EQUIVALENCE_INPUT_LIMITS, z as OpenAICompatibleOptimizerModel, zt as CompareOptimizationMethodsOptions } from "../skillopt-optimization-method-DJ3l4w8W.js";
|
|
8
|
+
import { A as ScalarDistribution, C as InsightReport, D as OutcomeCorrelationInsight, E as LiftInsight, O as Recommendation, S as FailureClusterInsight, T as JudgeInsight, b as ExecutionInsight, c as EvalRunGenerationSnapshot, g as TraceSpanEvent, j as TokenUsageInsight, k as ReleaseSummary, n as HostedTenant, o as EvalRunCellScore, s as EvalRunEvent, v as CostProvenanceSummary, w as InterRaterInsight, x as FailureClassTally, y as ExecutionErrorOutcomeCell } from "../client-BIyh1RCr.js";
|
|
8
9
|
import { i as InMemoryOutcomeStore, n as FileSystemOutcomeStore, o as OutcomeStore, r as FileSystemOutcomeStoreOptions, t as DeploymentOutcome } from "../outcome-store-BYHIuO0e.js";
|
|
9
|
-
import { a as summarizeExecution, i as analyzeRuns, n as ExecutionReport, r as SummarizeExecutionOptions, t as AnalyzeRunsOptions } from "../analyze-runs-
|
|
10
|
+
import { a as summarizeExecution, i as analyzeRuns, n as ExecutionReport, r as SummarizeExecutionOptions, t as AnalyzeRunsOptions } from "../analyze-runs-DMo3Lb_y.js";
|
|
10
11
|
import { AgentCandidateBenchmarkCellRef, AgentCandidateBenchmarkSuiteInputs, AgentCandidateBenchmarkTask, AgentCandidateBenchmarkTaskMaterial, AgentCandidateBundle, AgentCandidateEvaluationPolicy, AgentCandidateExperiment, AgentCandidateExperimentMaterial, AgentCandidateExperimentMeasurement, AgentImprovementCost, AgentImprovementMeasuredComparison, AgentProfileImprovementExperiment, AgentProfileImprovementExperimentMaterial, AgentProfileImprovementMeasuredComparison, AgentProfileImprovementMeasurement, AgentProfileImprovementRunCell, AgentProfileImprovementRunReceipt, AgentProfileImprovementSuiteInputs, AgentProfileImprovementTask, AgentProfileImprovementTaskMaterial, CandidateExecutionEvidence, Sha256Digest } from "@tangle-network/agent-interface";
|
|
11
12
|
//#region src/contract/self-improve.d.ts
|
|
12
13
|
interface SelfImproveBudget {
|
|
@@ -126,7 +127,7 @@ interface SelfImproveOptions<TScenario extends Scenario, TArtifact> {
|
|
|
126
127
|
* Candidate generator for this local generation loop.
|
|
127
128
|
* Required when `budget.generations` is greater than zero.
|
|
128
129
|
*/
|
|
129
|
-
proposer?: SurfaceProposer
|
|
130
|
+
proposer?: SurfaceProposer<ProposalFinding>;
|
|
130
131
|
/**
|
|
131
132
|
* Complete optimization method, such as official GEPA or SkillOpt.
|
|
132
133
|
* The method receives disjoint train and selection cases and never receives
|
|
@@ -196,9 +197,9 @@ interface SelfImproveOptions<TScenario extends Scenario, TArtifact> {
|
|
|
196
197
|
/** Free-form labels attached to the hosted event (env, branch, model id,
|
|
197
198
|
* etc.). Ignored when `hostedTenant` is unset. */
|
|
198
199
|
hostedLabels?: Record<string, string>;
|
|
199
|
-
/** Capture every artifact
|
|
200
|
-
*
|
|
201
|
-
* `'off'` to disable. Default: off. */
|
|
200
|
+
/** Capture every search artifact and judge score to this store.
|
|
201
|
+
* The store is output only and is never exposed to candidate generation.
|
|
202
|
+
* Pass `'off'` to disable. Default: off. */
|
|
202
203
|
labeledStore?: LabeledScenarioStore | 'off';
|
|
203
204
|
/** Capture-source tag for `labeledStore`. Default `'eval-run'`. */
|
|
204
205
|
captureSource?: 'production-trace' | 'eval-run' | 'manual' | 'red-team' | 'synthetic';
|
|
@@ -221,7 +222,7 @@ interface SelfImproveOptions<TScenario extends Scenario, TArtifact> {
|
|
|
221
222
|
analyzeGeneration?: RunOptimizationOptions<TScenario, TArtifact>['analyzeGeneration'];
|
|
222
223
|
/** Static findings forwarded to the proposer's `propose()` as `ctx.findings`
|
|
223
224
|
* (a findings-grounded proposer consumes them). Default: none. */
|
|
224
|
-
findings?:
|
|
225
|
+
findings?: ProposalFinding[];
|
|
225
226
|
/** Override how the WINNER is selected among coverage-complete candidates.
|
|
226
227
|
* Defaults to the scalar mean composite (historical behavior). A binary-with-
|
|
227
228
|
* replicates consumer whose ship-gate counts an instance resolved only when
|
|
@@ -1065,5 +1066,5 @@ interface FromOtelSpansOptions {
|
|
|
1065
1066
|
}
|
|
1066
1067
|
declare function fromOtelSpans(opts: FromOtelSpansOptions): RunRecord[];
|
|
1067
1068
|
//#endregion
|
|
1068
|
-
export { type AgentEvalAgent, type AgentEvalEvaluateOptions, type AgentEvalImproveOptions, type AgentProfileImprovementExperimentExecutionInput, type AgentProfileImprovementExperimentRun, type AgentTraceContributor, type AgentTraceContributorType, type AgentTraceConversation, type AgentTraceFile, type AgentTraceIndex, type AgentTraceRange, type AgentTraceRecord, type AnalystFinding, type AnalyzeRunsOptions, type AuthoringProvenance, type AxisEvidence, type AxisVerdict, type BuildEvidenceVectorOptions, type CampaignAggregates, type CampaignArtifactWriter, type CampaignCellFailureReceipt, type CampaignCellResult, type CampaignCostMeter, type CampaignResult, type CampaignStorage, type CampaignTraceWriter, type CandidateExperimentExecutionInput, type CandidateExperimentRun, type ChatClient, type CodeAgentSessionAction, type CodeAgentSessionActionKind, type CodeAgentSessionActionStatus, type CodeAgentSessionActionSurface, type CodeAgentSessionDiagnostic, type CodeAgentSessionExecutionReceipt, type CodeAgentSessionIntakeOptions, type CodeAgentSessionIntakeResult, type CodeAgentSessionMetrics, type CodeAgentSessionObservation, type CodeAgentSessionSource, type CodeAgentSessionTerminalStatus, type CodeSurface, type CompareAgentProfileImprovementExperimentOptions, type CompareCandidateExperimentOptions, type CompareOptimizationMethodsOptions, type ComparisonCost, type CostLedgerHandle, type CostProvenanceSummary, type CreateChatClientOpts, type DefaultAnalystRegistryOptions, type DefaultProductionGateCheck, type DefaultProductionGateOptions, type DefaultProductionRewardHackingOptions, type DefineAgentEvalOptions, type DefinedAgentEval, type DeploymentOutcome, type DispatchFn as Dispatch, type DispatchContext, type EvalCellScoreDelta, type EvalDimensionDelta, type EvalGenerationDiff, type EvalReportingSuiteInput, type EvalReportingSuiteOptions, type EvalReportingSuiteResult, type EvalRunDiff, type EvaluatePairedMeasurementsOptions, type EvidenceVector, type ExecutionErrorOutcomeCell, type ExecutionInsight, type ExecutionReport, type ExternalOptimizationExample, type ExternalTextEvaluationResponse, type ExternalTextOptimizationMethodConfig, type ExternalTextOptimizerContext, type ExternalTextOptimizerResult, type FailureClassTally, type FailureClusterInsight, type FeedbackTableMeta, type FeedbackTableRow, FileSystemOutcomeStore, type FileSystemOutcomeStoreOptions, type FromFeedbackTableOptions, type FromFeedbackTableResult, type FromOtelSpansOptions, type FromRunRecordDirOptions, type FromRunRecordDirResult, type Gate, type GateCheckStatus, type GateContext, type GateContribution, type GateDecision, type GateResult, type GenerationCandidate, type GenerationRecord, type GepaAdaptiveEngineRun, type GepaEngineOptions, type GepaEngineRun, type GepaOptimizationMethodConfig, type GepaOptimizationRecipe, type GepaRunnerCommand, type HeldOutGateOptions, type HostedTenant, InMemoryOutcomeStore, type InsightReport, type InterRaterInsight, type JudgeConfig, type JudgeDimension, type JudgeInsight, type JudgeScore, type LiftInsight, type LlmJudgeDimension, type LlmJudgeOptions, type MutableSurface, type ObjectiveSource, type OpenAICompatibleOptimizerModel, type OptimizationMethod, type OptimizationMethodComparison, type OptimizationMethodInput, type OptimizationMethodProvenance, type OptimizationMethodResult, type OptimizationPackageSource, type OptimizationTokenUsage, type OptimizerConfig, type OptimizerModelBudget, type OutcomeCorrelationInsight, type OutcomeStore, type PairedMeasurement, type PairedMeasurementAdapter, type PairedMeasurementEvaluation, type ParetoSignificanceGateOptions, type ParsedCodeAgentJsonl, type PartitionByAuthoringModelResult, type PromotionObjective, type PromotionPolicy, REFERENCE_EQUIVALENCE_INPUT_LIMITS, REFERENCE_EQUIVALENCE_JUDGE_VERSION, type Recommendation, type ReferenceEquivalenceJudgeInput, type ReferenceEquivalenceJudgeOptions, type ReferenceEquivalenceJudgeResult, type ReferenceEquivalenceScenario, type ReleaseSummary, type RunAgentProfileImprovementExperimentOptions, type RunCampaignOptions, type RunCandidateExperimentOptions, type RunEvalOptions, type RunImprovementLoopOptions, type RunImprovementLoopResult, type RunRecordRejection, type ScalarDistribution, type Scenario, type SealAgentProfileImprovementSuiteOptions, type SealCandidateBenchmarkSuiteOptions, type SelfImproveBudget, type SelfImproveOptions, type SelfImproveProgressEvent, type SelfImproveResult, SelfImproveRunError, type SessionScript, type SkillOptOptimizationMethodConfig, type SkillOptRunnerCommand, type SkillOptTrainerConfig, type SummarizeExecutionOptions, type SurfaceProposer, type TokenUsageInsight, analyzeRuns, buildDefaultAnalystRegistry, buildEvidenceVector, campaignSplitDigest, compareOptimizationMethods, composeGate, createChatClient, createReferenceEquivalenceJudge, defaultProductionGate, defineAgentEval, diffGenerations, diffRunBaselineToWinner, diffRuns, evalReportingSuite, evaluatePairedMeasurements, externalTextOptimizationMethod, fromClaudeCodeSession, fromCodexSession, fromFeedbackTable, fromKimiCodeSession, fromOpenCodeSession, fromOtelSpans, fromPiSession, fromPigraphSession, fromRunRecordDir, fsCampaignStorage, gepaOptimizationMethod, heldOutGate, inMemoryCampaignStorage, llmJudge, measuredComparisonFromAgentProfileImprovementExperiment, measuredComparisonFromCandidateExperiment, observeCodeAgentSession, paretoPolicy, paretoSignificanceGate, parseAgentTrace, parseCodeAgentJsonl, partitionRunsByAuthoringModel, runAgentProfileImprovementExperiment, runCampaign, runCandidateExperiment, runEval, runImprovementLoop, runReferenceEquivalenceJudge, sealAgentProfileImprovementExperiment, sealAgentProfileImprovementSuite, sealAgentProfileImprovementTask, sealCandidateBenchmarkSuite, sealCandidateBenchmarkTask, sealCandidateExperiment, selfImprove, skillOptOptimizationMethod, summarizeExecution, verifyAgentProfileImprovementExperiment, verifyAgentProfileImprovementExperimentComparison, verifyAgentProfileImprovementSuiteInputs, verifyAgentProfileImprovementTask, verifyCandidateBenchmarkSuite, verifyCandidateBenchmarkSuiteInputs, verifyCandidateBenchmarkTask, verifyCandidateExperiment, verifyCandidateExperimentComparison };
|
|
1069
|
+
export { type AgentEvalAgent, type AgentEvalEvaluateOptions, type AgentEvalImproveOptions, type AgentProfileImprovementExperimentExecutionInput, type AgentProfileImprovementExperimentRun, type AgentTraceContributor, type AgentTraceContributorType, type AgentTraceConversation, type AgentTraceFile, type AgentTraceIndex, type AgentTraceRange, type AgentTraceRecord, type AnalystFinding, type AnalyzeRunsOptions, type AuthoringProvenance, type AxisEvidence, type AxisVerdict, type BuildEvidenceVectorOptions, type CampaignAggregates, type CampaignArtifactWriter, type CampaignCellFailureReceipt, type CampaignCellResult, type CampaignCostMeter, type CampaignResult, type CampaignStorage, type CampaignTraceWriter, type CandidateExperimentExecutionInput, type CandidateExperimentRun, type ChatClient, type CodeAgentSessionAction, type CodeAgentSessionActionKind, type CodeAgentSessionActionStatus, type CodeAgentSessionActionSurface, type CodeAgentSessionDiagnostic, type CodeAgentSessionExecutionReceipt, type CodeAgentSessionIntakeOptions, type CodeAgentSessionIntakeResult, type CodeAgentSessionMetrics, type CodeAgentSessionObservation, type CodeAgentSessionSource, type CodeAgentSessionTerminalStatus, type CodeSurface, type CompareAgentProfileImprovementExperimentOptions, type CompareCandidateExperimentOptions, type CompareOptimizationMethodsOptions, type ComparisonCost, type CostLedgerHandle, type CostProvenanceSummary, type CreateChatClientOpts, type DefaultAnalystRegistryOptions, type DefaultProductionGateCheck, type DefaultProductionGateOptions, type DefaultProductionRewardHackingOptions, type DefineAgentEvalOptions, type DefinedAgentEval, type DeploymentOutcome, type DispatchFn as Dispatch, type DispatchContext, type EvalCellScoreDelta, type EvalDimensionDelta, type EvalGenerationDiff, type EvalReportingSuiteInput, type EvalReportingSuiteOptions, type EvalReportingSuiteResult, type EvalRunDiff, type EvaluatePairedMeasurementsOptions, type EvidenceVector, type ExecutionErrorOutcomeCell, type ExecutionInsight, type ExecutionReport, type ExternalOptimizationExample, type ExternalTextEvaluationResponse, type ExternalTextOptimizationMethodConfig, type ExternalTextOptimizerContext, type ExternalTextOptimizerResult, type FailureClassTally, type FailureClusterInsight, type FeedbackTableMeta, type FeedbackTableRow, FileSystemOutcomeStore, type FileSystemOutcomeStoreOptions, type FromFeedbackTableOptions, type FromFeedbackTableResult, type FromOtelSpansOptions, type FromRunRecordDirOptions, type FromRunRecordDirResult, type Gate, type GateCheckStatus, type GateContext, type GateContribution, type GateDecision, type GateResult, type GenerationCandidate, type GenerationRecord, type GepaAdaptiveEngineRun, type GepaEngineOptions, type GepaEngineRun, type GepaOptimizationMethodConfig, type GepaOptimizationRecipe, type GepaRunnerCommand, type HeldOutGateOptions, type HostedTenant, InMemoryOutcomeStore, type InsightReport, type InterRaterInsight, type JudgeConfig, type JudgeDimension, type JudgeInsight, type JudgeScore, type LiftInsight, type LlmJudgeDimension, type LlmJudgeOptions, type MutableSurface, type ObjectiveSource, type OpenAICompatibleOptimizerModel, type OptimizationMethod, type OptimizationMethodComparison, type OptimizationMethodInput, type OptimizationMethodProvenance, type OptimizationMethodResult, type OptimizationPackageSource, type OptimizationTokenUsage, type OptimizerConfig, type OptimizerModelBudget, type OutcomeCorrelationInsight, type OutcomeStore, type PairedMeasurement, type PairedMeasurementAdapter, type PairedMeasurementEvaluation, type ParetoSignificanceGateOptions, type ParsedCodeAgentJsonl, type PartitionByAuthoringModelResult, type PromotionObjective, type PromotionPolicy, type ProposalFinding, type ProposalFindingOrigin, REFERENCE_EQUIVALENCE_INPUT_LIMITS, REFERENCE_EQUIVALENCE_JUDGE_VERSION, type Recommendation, type ReferenceEquivalenceJudgeInput, type ReferenceEquivalenceJudgeOptions, type ReferenceEquivalenceJudgeResult, type ReferenceEquivalenceScenario, type ReleaseSummary, type RunAgentProfileImprovementExperimentOptions, type RunCampaignOptions, type RunCandidateExperimentOptions, type RunEvalOptions, type RunImprovementLoopOptions, type RunImprovementLoopResult, type RunRecordRejection, type ScalarDistribution, type Scenario, type SealAgentProfileImprovementSuiteOptions, type SealCandidateBenchmarkSuiteOptions, type SelfImproveBudget, type SelfImproveOptions, type SelfImproveProgressEvent, type SelfImproveResult, SelfImproveRunError, type SessionScript, type SkillOptOptimizationMethodConfig, type SkillOptRunnerCommand, type SkillOptTrainerConfig, type SummarizeExecutionOptions, type SurfaceProposer, type TokenUsageInsight, analyzeRuns, buildDefaultAnalystRegistry, buildEvidenceVector, campaignSplitDigest, compareOptimizationMethods, composeGate, createChatClient, createReferenceEquivalenceJudge, defaultProductionGate, defineAgentEval, diffGenerations, diffRunBaselineToWinner, diffRuns, evalReportingSuite, evaluatePairedMeasurements, externalTextOptimizationMethod, fromClaudeCodeSession, fromCodexSession, fromFeedbackTable, fromKimiCodeSession, fromOpenCodeSession, fromOtelSpans, fromPiSession, fromPigraphSession, fromRunRecordDir, fsCampaignStorage, gepaOptimizationMethod, heldOutGate, inMemoryCampaignStorage, llmJudge, makeProposalFinding, measuredComparisonFromAgentProfileImprovementExperiment, measuredComparisonFromCandidateExperiment, observeCodeAgentSession, paretoPolicy, paretoSignificanceGate, parseAgentTrace, parseCodeAgentJsonl, partitionRunsByAuthoringModel, runAgentProfileImprovementExperiment, runCampaign, runCandidateExperiment, runEval, runImprovementLoop, runReferenceEquivalenceJudge, sealAgentProfileImprovementExperiment, sealAgentProfileImprovementSuite, sealAgentProfileImprovementTask, sealCandidateBenchmarkSuite, sealCandidateBenchmarkTask, sealCandidateExperiment, selfImprove, skillOptOptimizationMethod, summarizeExecution, verifyAgentProfileImprovementExperiment, verifyAgentProfileImprovementExperimentComparison, verifyAgentProfileImprovementSuiteInputs, verifyAgentProfileImprovementTask, verifyCandidateBenchmarkSuite, verifyCandidateBenchmarkSuiteInputs, verifyCandidateBenchmarkTask, verifyCandidateExperiment, verifyCandidateExperimentComparison };
|
|
1069
1070
|
//# sourceMappingURL=index.d.ts.map
|
|
@@ -1 +1 @@
|
|
|
1
|
-
{"version":3,"file":"index.d.ts","names":[],"sources":["../../src/contract/self-improve.ts","../../src/contract/define-agent-eval.ts","../../src/contract/measured-comparison.ts","../../src/contract/profile-measured-comparison.ts","../../src/contract/intake/run-record-dir.ts","../../src/contract/eval-reporting-suite.ts","../../src/contract/diff.ts","../../src/contract/intake/agent-trace.ts","../../src/contract/intake/code-agent-observation.ts","../../src/contract/intake/code-agent-session.ts","../../src/contract/intake/feedback-table.ts","../../src/contract/intake/otel-spans.ts"],"mappings":"
|
|
1
|
+
{"version":3,"file":"index.d.ts","names":[],"sources":["../../src/contract/self-improve.ts","../../src/contract/define-agent-eval.ts","../../src/contract/measured-comparison.ts","../../src/contract/profile-measured-comparison.ts","../../src/contract/intake/run-record-dir.ts","../../src/contract/eval-reporting-suite.ts","../../src/contract/diff.ts","../../src/contract/intake/agent-trace.ts","../../src/contract/intake/code-agent-observation.ts","../../src/contract/intake/code-agent-session.ts","../../src/contract/intake/feedback-table.ts","../../src/contract/intake/otel-spans.ts"],"mappings":";;;;;;;;;;;;UA+DiB;;;EAGf;;;;EAIA;;EAEA;;EAEA;;;EAGA;;;EAGA;;;;EAIA;;EAEA,mBAAmB;;;;;;;;;EASnB;;EAEA;;;;;EAKA;;KAGU;EACN;EAA0B;;EAC1B;EAA4B;EAAuB;;EACnD;EAA4B;EAAe;;EAC3C;EAA8B;EAAe;EAAuB;;EAGpE;EAAsB;EAAkB;;EACxC;EAAyB;EAAW;EAAY;EAAa;;UAElD,mBAAmB,kBAAkB,UAAU;;;;;;;;;;;;;;EAc9D,QAAQ,SAAS,gBAAgB,UAAU,WAAW,KAAK,oBAAoB,QAAQ;;;;;;;EAQvF;;;EAIA,WAAW;;;EAIX,OAAO,YAAY,WAAW;;;EAI9B,iBAAiB;;EAGjB,SAAS;;;;;;;;;;;;EAaT,sBAAsB,gCAAgC,WAAW;;;;;EAMjE,WAAW,gBAAgB;;;;;;EAO3B,SAAS,mBAAmB,WAAW;;;EAIvC,qBAAqB;;;EAIrB,OAAO,KAAK,WAAW;;;;;;;;EASvB,cAAc,eAAe,gBAAgB,iBAAiB,mBAAmB;;;;EAKjF,UAAU;;;;EAKV;;;EAIA,gBAAgB,QAAQ;;;;;EAMxB,iBAAiB;IACf,UAAU;IACV;IACA;;;;EAKF;;;EAIA,cAAc,OAAO;;;;EAKrB;EACA;EACA;;;;;;;;;;;;;;EAeA,eAAe;;;EAIf,eAAe;;;;EAKf,eAAe;;EAGf;;;;;;;;EASA;;;;;;;;;EAUA,oBAAoB,uBAAuB,WAAW;;;EAItD,WAAW;;;;;;;EAQX,mBAAmB,uBAAuB,WAAW;;UAGtC,kBAAkB,kBAAkB,UAAU;;;;EAI7D;IACE;IACA,aAAa;;;;;EAKf;IACE;IACA,aAAa;IACb,SAAS;;;IAGT;;;;IAIA;;;;;;EAMF;;;EAGA;;;;EAIA,YAAY;;EAEZ;;;EAGA;;EAEA;;EAEA;;EAEA,MAAM;;;EAGN,UAAU;;EAEV;IACE;IACA,MAAM;IACN;IACA,aAAa;;;;;;;;EAQf,SAAS;;;;;EAKT,QAAQ;;;;;;EAMR,KAAK,yBAAyB,WAAW;;;cAI9B,4BAA4B;WAC9B,MAAM;WACN,UAAU;EAEnB,YAAY,gBAAgB,QAAQ;;;;;;;;;;;;;;;;;;;;;;;;;;;;iBA+LhB,YAAY,kBAAkB,UAAU,WAC5D,MAAM,mBAAmB,WAAW,aACnC,QAAQ,kBAAkB,WAAW;;;KCviB5B,eAAe,kBAAkB,UAAU,cACrD,SAAS,gBACT,UAAU,WACV,KAAK,oBACF,QAAQ;KAED,uBAAuB,kBAAkB,UAAU,aAAa,mBAC1E,WACA;UAGe,yBAAyB,kBAAkB,UAAU,mBAC5D,KACN,eAAe,WAAW;;EAI5B,YAAY;;EAEZ,UAAU;;EAEV,QAAQ,eAAe,WAAW;;EAElC,QAAQ,YAAY,WAAW;;EAE/B,SAAS,YAAY,WAAW;;EAEhC;;KAGU,wBAAwB,kBAAkB,UAAU,aAAa,KAC3E,QAAQ,mBAAmB,WAAW;EAGtC,SAAS,QAAQ;EACjB,eAAe,QAAQ;;UAGR,iBAAiB,kBAAkB,UAAU;;WAEnD,oBAAoB;;WAEpB,iBAAiB;;;;;EAK1B,SACE,OAAO,yBAAyB,WAAW,aAC1C,QAAQ,eAAe,WAAW;;;;;;;EAOrC,QACE,OAAO,wBAAwB,WAAW,aACzC,QAAQ,kBAAkB,WAAW;;;;;;;;;iBAU1B,gBAAgB,kBAAkB,UAAU,WAC1D,UAAU,uBAAuB,WAAW,aAC3C,iBAAiB,WAAW;;;UCxDd;EACf,QAAQ,gCAAgC;EACxC;EACA;;UAGe;EACf,YAAY;EACZ;EACA,QAAQ;EACR,MAAM;EACN,eAAe;EACf;EACA,SAAS;;UAGM;EACf,YAAY;EACZ,QAAQ,OAAO,oCAAoC,QAAQ;;EAE3D;;EAEA,aAAa;EACb,SAAS;;UAGM;EACf,cAAc;EACd;IACE;IACA,MAAM;;;UAIO;EACf,YAAY;EACZ,cAAc;EACd;IACE;IACA,MAAM;;EAER,aAAa;EACb;EACA,YAAY;EACZ;EACA,WAAW;;;UAII,kBAAkB;EACjC;EACA,UAAU;EACV,WAAW;;;UAII,yBAAyB;EACxC,MAAM,KAAK;EACX,WAAW,KAAK;IAAkB;IAAc;;EAChD,QAAQ,KAAK;EACb,eAAe,KAAK,OAAO;EAC3B,UAAU,KAAK;EACf,UAAU,KAAK;EACf,OAAO,KAAK;;UAGG,kCAAkC;EACjD,uBAAuB,kBAAkB;EACzC,QAAQ;EACR,SAAS,yBAAyB;;EAElC;;EAEA,kBAAkB;;EAElB,kBAAkB;;;KAIR,8BAA8B,KACxC;EAGA,iBAAiB;EACjB,WAAW;EACX;;;iBAIc,2BACd,UAAU,sCACT;;iBAQa,4BACd,SAAS,qCACR;;iBAiBa,wBACd,UAAU,mCACT;iBAQa,0BAA0B,iBAAiB;;iBAarC,uBACpB,SAAS,gCACR,QAAQ;;;;;;;;iBAkEK,2BAA2B,MACzC,SAAS,kCAAkC,QAC1C;;iBA+Pa,0CACd,SAAS,oCACR;;iBAuEa,oCACd,iBACC;iBAkHa,6BAA6B,iBAAiB;iBAM9C,oCACd,iBACC;iBAkBa,8BAA8B;;;;;;;;;;UC7qB7B;EACf,aAAa;EACb,QAAQ,gCAAgC;EACxC;EACA;;UAGe;EACf,YAAY;EACZ;EACA,aAAa;EACb,MAAM;EACN,SAAS;EACT;EACA,SAAS;;UAGM;EACf,YAAY;EACZ,QACE,OAAO,kDACN,QAAQ;;EAEX;;EAEA,aAAa;EACb,SAAS;;UAGM;EACf,cAAc;EACd;IACE;IACA,MAAM;;;UAIO;EACf,YAAY;EACZ,cAAc;EACd;IACE;IACA,MAAM;;EAER,aAAa;EACb;EACA,YAAY;EACZ;EACA,WAAW;;;iBAIG,gCACd,UAAU,sCACT;;iBAQa,iCACd,SAAS,0CACR;;iBAqBa,sCACd,UAAU,4CACT;iBAOa,kCAAkC,iBAAiB;iBAInD,yCACd,iBACC;iBAIa,wCACd,iBACC;;;;;;iBASmB,qCACpB,SAAS,8CACR,QAAQ;;iBA0CK,wDACd,SAAS,kDACR;;iBAuEa,kDACd,iBACC;;;;UCnPc;;EAEf;;EAEA;;EAEA;;UAGe;;;;;;EAMf;;;;;;;EAOA,WAAW;;;;;;EAMX;;UAGe;;EAEf,MAAM;;EAEN,UAAU;;EAEV;;;;;;;;;;iBAkBoB,iBACpB,cACA,UAAS,0BACR,QAAQ;;;;;KC7CC,0BAA0B;UAErB;;;;;EAKf,UAAU,KAAK;;EAEf,OAAO;;;;;;;;;;EAUP;;;;UAKe;;;EAGf,QAAQ;;EAER;;IAEE;;IAEA;;;IAGA;;IAEA;;;IAGA,UAAU;;;EAGZ;;;;;;;iBAUoB,mBACpB,OAAO,yBACP,UAAS,4BACR,QAAQ;;;;;;UC5DM;EACf;EACA;EACA;;;UAIe;EACf;EACA;EACA;EACA;EACA;;;EAGA,YAAY,eAAe,eAAe;;;;UAK3B;EACf;EACA;EACA;EACA;EACA;;EAEA,SAAS;;EAET,SAAS;;EAET,OAAO;;EAEP;EACA;EACA;EACA;EACA;EACA;EACA;EACA;EACA;;;;;UAMe;EACf;EACA;EACA;EACA;EACA,oBAAoB;EACpB,mBAAmB;EACnB;EACA;EACA;EACA;EACA;EACA;EACA;EACA;EACA;;EAEA,cAAc;;;;EAId,aAAa;;;;;;;;;iBAoDC,gBACd,QAAQ,2BACR,OAAO,4BACN;;;;;;iBAqEa,SAAS,QAAQ,cAAc,OAAO,eAAe;;;;;;;iBAsCrD,wBAAwB,KAAK,eAAe;;;KC1OhD;UAEK;EACf,MAAM;;EAEN;;UAGe;EACf;EACA;EACA;;;EAGA,cAAc;;UAGC;EACf;EACA,cAAc;EACd,QAAQ;;UAGO;EACf;EACA,eAAe;;UAGA;EACf;EACA;EACA;EACA;IAAQ;IAAc;;EACtB;IAAS;IAAe;;EACxB,OAAO;;;;UAOQ;EACf;;EAEA;;EAEA;EACA;EACA;;EAEA;;EAEA;;KAGU,kBAAkB,YAAY;;;;;;iBAW1B,gBAAgB,SAAS,qBAAqB;UAmE7C;;;;EAIf,SAAS,YAAY;;;EAGrB,cAAc;;;;;;;;iBASA,8BACd,MAAM,aACN,OAAO,kBACN;;;KCjLS;KAEA;KAEA;KAEA;KASA;UAEK;EACf;EACA;EACA;;UAGe;EACf;EACA;EACA,MAAM;EACN,SAAS;EACT;EACA,QAAQ;EACR;EACA;EACA,UAAU;;UAGK;EACf,QAAQ;EACR;EACA;EACA;IACE,QAAQ;IACR;;EAEF,SAAS;;UAGM;EACf,QAAQ;EACR;EACA;EACA,YAAY;;;;;;;iBAeE,wBACd,SAAS,iCACR;;;UCrCc;EACf;EACA;;UAGe;EACf;EACA;EACA;EACA;EACA;EACA;EACA;EACA;EACA;EACA;EACA;EACA;EACA;EACA;EACA;EACA;EACA;EACA;EACA;EACA;EACA;EACA;EACA;EACA;EACA;EACA;EACA;EACA;EACA;EACA;EACA;EACA;EACA;EACA;EACA;EACA;EACA;EACA;EACA;;UAGe;EACf,QAAQ;EACR;EACA;EACA;EACA;EACA;EACA;EACA;EACA;EACA;EACA,WAAW;EACX;;UAGe;EACf,MAAM;EACN,aAAa;EACb,SAAS;EACT,cAAc;;UAGC;EACf;EACA;EACA;EACA;EACA;EACA;EACA,WAAW;EACX;EACA;EACA;EACA;EACA;EACA;;;EAGA,iBAAiB;;;EAGjB,YAAY;;iBAGE,oBAAoB,gBAAgB;iBAepC,iBACd,SAAS,gCACR;iBAIa,sBACd,SAAS,gCACR;iBAIa,oBACd,SAAS,gCACR;iBAIa,oBACd,SAAS,gCACR;iBAIa,cACd,SAAS,gCACR;cAIU,2BAAkB;;;UCpJd;;;EAGf;;EAEA;;;EAGA;;;EAGA,WAAW;;UAGI;EACf;;;EAGA;;EAEA;;EAEA;;EAEA;;EAEA;;EAEA;;EAEA;;EAEA;;EAEA;;;EAGA,WAAW;;;EAGX,SAAS;;UAGM;;EAEf,SAAS;;;EAGT,OAAO;;;;EAIP;IAAU;IAAa;;;;;EAIvB;;UAGe;EACf,MAAM;;;EAGN,aAAa;IAAQ;IAAe;IAAe;;;iBAGrC,kBAAkB,MAAM,2BAA2B;;;UCvBlD;EACf,OAAO;;EAEP,eAAe;;EAEf;;;;;;EAMA,eAAe,eAAe,gBAAgB;;iBAGhC,cAAc,MAAM,uBAAuB"}
|
package/dist/contract/index.js
CHANGED
|
@@ -1,30 +1,23 @@
|
|
|
1
1
|
import { s as ValidationError } from "../errors-8YnH8WlF.js";
|
|
2
|
-
import {
|
|
2
|
+
import { F as createChatClient, t as buildDefaultAnalystRegistry } from "../default-registry-IjYs7T8l.js";
|
|
3
3
|
import { g as isModelPriced, i as CostLedger, m as estimateCost } from "../cost-ledger-BrJxbrMy.js";
|
|
4
4
|
import { LLM_MODEL_ATTR_KEYS, SPAN_KIND_ATTR_KEYS } from "../trace-attributes.js";
|
|
5
5
|
import { b as classifyOtlpSpanRole, x as isOtlpModelCall } from "../tools-BmuN627J.js";
|
|
6
|
+
import { l as makeProposalFinding } from "../proposal-findings-DCawte-y.js";
|
|
6
7
|
import { r as mapConcurrentRange } from "../concurrency-MUjT7VjM.js";
|
|
7
|
-
import { B as surfaceContentHash, Ct as llmJudge, M as compareOptimizationMethods, O as composeGate, Q as inMemoryCampaignStorage, S as defaultProductionGate, T as heldoutSignificance, V as surfaceHash, X as createRunCostLedger, Y as runCampaign, Z as fsCampaignStorage, _ as buildEvidenceVector, a as emitLoopProvenance, b as powerPreflight, ct as campaignSplitDigest, d as runImprovementLoop, dt as REFERENCE_EQUIVALENCE_INPUT_LIMITS, ft as REFERENCE_EQUIVALENCE_JUDGE_VERSION, g as gepaOptimizationMethod, j as assertOptimizationResult, k as externalTextOptimizationMethod, mt as runReferenceEquivalenceJudge, o as loopProvenanceArgsFromResult, p as runEval, pt as createReferenceEquivalenceJudge, rt as resolveRunDir, t as skillOptOptimizationMethod, v as paretoPolicy, x as heldOutGate, y as paretoSignificanceGate } from "../skillopt-optimization-method-
|
|
8
|
-
import {
|
|
8
|
+
import { B as surfaceContentHash, Ct as llmJudge, M as compareOptimizationMethods, O as composeGate, Q as inMemoryCampaignStorage, S as defaultProductionGate, T as heldoutSignificance, V as surfaceHash, X as createRunCostLedger, Y as runCampaign, Z as fsCampaignStorage, _ as buildEvidenceVector, a as emitLoopProvenance, b as powerPreflight, ct as campaignSplitDigest, d as runImprovementLoop, dt as REFERENCE_EQUIVALENCE_INPUT_LIMITS, ft as REFERENCE_EQUIVALENCE_JUDGE_VERSION, g as gepaOptimizationMethod, j as assertOptimizationResult, k as externalTextOptimizationMethod, mt as runReferenceEquivalenceJudge, o as loopProvenanceArgsFromResult, p as runEval, pt as createReferenceEquivalenceJudge, rt as resolveRunDir, t as skillOptOptimizationMethod, v as paretoPolicy, x as heldOutGate, y as paretoSignificanceGate } from "../skillopt-optimization-method-BY6vKLJB.js";
|
|
9
|
+
import { C as pairedBootstrap } from "../statistics-RwRNu2__.js";
|
|
9
10
|
import { i as parseRunRecordSafe, r as modelHasSnapshot } from "../run-record-BIwU2wdV.js";
|
|
10
11
|
import { a as recordAggregateMeasurements, i as readTaskFailureLabels, o as summarizeExecutionMeasurements, s as summarizeTraceErrors, t as extractUsage } from "../extract-usage-2j25whHw.js";
|
|
11
|
-
import { n as summarizeExecution, t as analyzeRuns } from "../analyze-runs-
|
|
12
|
-
import { a as campaignCellExecutionEvidence, c as campaignCellToRunRecord, o as campaignCellJudgeDimensions, s as campaignCellTaskScore } from "../reward-hacking-
|
|
12
|
+
import { n as summarizeExecution, t as analyzeRuns } from "../analyze-runs-qk8op0tN.js";
|
|
13
|
+
import { a as campaignCellExecutionEvidence, c as campaignCellToRunRecord, o as campaignCellJudgeDimensions, s as campaignCellTaskScore } from "../reward-hacking-DCdRK9TY.js";
|
|
13
14
|
import { n as InMemoryOutcomeStore, t as FileSystemOutcomeStore } from "../outcome-store-ChBKlTd_.js";
|
|
14
|
-
import { t as createHostedClient } from "../client-
|
|
15
|
+
import { t as createHostedClient } from "../client-LIuo-KPv.js";
|
|
15
16
|
import { dirname, join } from "node:path";
|
|
16
17
|
import { createHash } from "node:crypto";
|
|
17
18
|
import { mkdir, readFile, readdir, stat, writeFile } from "node:fs/promises";
|
|
18
19
|
import { agentCandidateBenchmarkSuiteSchema, agentCandidateBenchmarkTaskSchema, agentCandidateBundleSchema, agentCandidateEvaluationPolicySchema, agentCandidateExperimentSchema, agentImprovementMeasuredComparisonSchema, agentProfileImprovementExperimentSchema, agentProfileImprovementMeasuredComparisonSchema, agentProfileImprovementRunCellSchema, agentProfileImprovementRunReceiptSchema, agentProfileImprovementSuiteInputsSchema, agentProfileImprovementSuiteSchema, agentProfileImprovementTaskSchema, candidateExecutionEvidenceSchema, canonicalCandidateDigest, canonicalCandidateJson, numbersApproximatelyEqual, omitTopLevelDigest } from "@tangle-network/agent-interface";
|
|
19
20
|
//#region src/contract/self-improve.ts
|
|
20
|
-
/**
|
|
21
|
-
* Run one complete improvement job.
|
|
22
|
-
*
|
|
23
|
-
* A caller-owned `proposer` can generate candidates across local generations.
|
|
24
|
-
* An external `method`, such as official GEPA or SkillOpt, owns its complete
|
|
25
|
-
* search and returns one candidate. Both paths remeasure the selected candidate
|
|
26
|
-
* against cases that candidate generation never receives.
|
|
27
|
-
*/
|
|
28
21
|
/** Failed self-improvement run with an immutable receipt snapshot. */
|
|
29
22
|
var SelfImproveRunError = class extends Error {
|
|
30
23
|
cost;
|
|
@@ -827,7 +820,7 @@ function evaluatePairedMeasurements(options) {
|
|
|
827
820
|
delta: overall.delta,
|
|
828
821
|
lowerBound: overall.confidenceInterval.lower,
|
|
829
822
|
deltaThreshold,
|
|
830
|
-
minProductiveRuns,
|
|
823
|
+
minProductiveRuns: significance.minimumRequired,
|
|
831
824
|
confidence,
|
|
832
825
|
sharedScorerChannel: options.sharedScorerChannel
|
|
833
826
|
});
|
|
@@ -878,8 +871,9 @@ function evaluatePairedMeasurements(options) {
|
|
|
878
871
|
}
|
|
879
872
|
];
|
|
880
873
|
const shipped = checks.every((check) => check.passed);
|
|
874
|
+
const hardFailure = incompleteRuns.length > 0 || failedCandidateResults.length > 0 || regressions.length > 0 || missingCriticalDimensions.length > 0 || !budgetPassed;
|
|
881
875
|
const reasons = [
|
|
882
|
-
...significance.significant ? [] : [significance.fewRuns ? `only ${significance.n} paired runs; ${
|
|
876
|
+
...significance.significant ? [] : [significance.fewRuns ? `only ${significance.n} paired runs; ${significance.minimumRequired} required` : `paired interval lower bound ${significance.bootstrap.low} did not clear ${deltaThreshold}`],
|
|
883
877
|
...powerSufficient || significance.fewRuns ? [] : [power.reason],
|
|
884
878
|
...regressions.length === 0 ? [] : [`critical dimensions regressed: ${regressions.map((entry) => entry.name).join(", ")}`],
|
|
885
879
|
...missingCriticalDimensions.length === 0 ? [] : [`critical dimensions missing: ${missingCriticalDimensions.join(", ")}`],
|
|
@@ -896,7 +890,7 @@ function evaluatePairedMeasurements(options) {
|
|
|
896
890
|
},
|
|
897
891
|
objectives,
|
|
898
892
|
decision: {
|
|
899
|
-
outcome: shipped ? "ship" : significance.fewRuns || !powerSufficient ? "need_more_work" : "hold",
|
|
893
|
+
outcome: shipped ? "ship" : hardFailure ? "hold" : significance.fewRuns || !powerSufficient ? "need_more_work" : "hold",
|
|
900
894
|
reasons: reasons.length > 0 ? reasons : ["all measured checks passed"],
|
|
901
895
|
contributingChecks: checks
|
|
902
896
|
},
|
|
@@ -3729,6 +3723,6 @@ function collectNumericAttrs(spans) {
|
|
|
3729
3723
|
return raw;
|
|
3730
3724
|
}
|
|
3731
3725
|
//#endregion
|
|
3732
|
-
export { FileSystemOutcomeStore, InMemoryOutcomeStore, REFERENCE_EQUIVALENCE_INPUT_LIMITS, REFERENCE_EQUIVALENCE_JUDGE_VERSION, SelfImproveRunError, analyzeRuns, buildDefaultAnalystRegistry, buildEvidenceVector, campaignSplitDigest, compareOptimizationMethods, composeGate, createChatClient, createReferenceEquivalenceJudge, defaultProductionGate, defineAgentEval, diffGenerations, diffRunBaselineToWinner, diffRuns, evalReportingSuite, evaluatePairedMeasurements, externalTextOptimizationMethod, fromClaudeCodeSession, fromCodexSession, fromFeedbackTable, fromKimiCodeSession, fromOpenCodeSession, fromOtelSpans, fromPiSession, fromPigraphSession, fromRunRecordDir, fsCampaignStorage, gepaOptimizationMethod, heldOutGate, inMemoryCampaignStorage, llmJudge, measuredComparisonFromAgentProfileImprovementExperiment, measuredComparisonFromCandidateExperiment, observeCodeAgentSession, paretoPolicy, paretoSignificanceGate, parseAgentTrace, parseCodeAgentJsonl, partitionRunsByAuthoringModel, runAgentProfileImprovementExperiment, runCampaign, runCandidateExperiment, runEval, runImprovementLoop, runReferenceEquivalenceJudge, sealAgentProfileImprovementExperiment, sealAgentProfileImprovementSuite, sealAgentProfileImprovementTask, sealCandidateBenchmarkSuite, sealCandidateBenchmarkTask, sealCandidateExperiment, selfImprove, skillOptOptimizationMethod, summarizeExecution, verifyAgentProfileImprovementExperiment, verifyAgentProfileImprovementExperimentComparison, verifyAgentProfileImprovementSuiteInputs, verifyAgentProfileImprovementTask, verifyCandidateBenchmarkSuite, verifyCandidateBenchmarkSuiteInputs, verifyCandidateBenchmarkTask, verifyCandidateExperiment, verifyCandidateExperimentComparison };
|
|
3726
|
+
export { FileSystemOutcomeStore, InMemoryOutcomeStore, REFERENCE_EQUIVALENCE_INPUT_LIMITS, REFERENCE_EQUIVALENCE_JUDGE_VERSION, SelfImproveRunError, analyzeRuns, buildDefaultAnalystRegistry, buildEvidenceVector, campaignSplitDigest, compareOptimizationMethods, composeGate, createChatClient, createReferenceEquivalenceJudge, defaultProductionGate, defineAgentEval, diffGenerations, diffRunBaselineToWinner, diffRuns, evalReportingSuite, evaluatePairedMeasurements, externalTextOptimizationMethod, fromClaudeCodeSession, fromCodexSession, fromFeedbackTable, fromKimiCodeSession, fromOpenCodeSession, fromOtelSpans, fromPiSession, fromPigraphSession, fromRunRecordDir, fsCampaignStorage, gepaOptimizationMethod, heldOutGate, inMemoryCampaignStorage, llmJudge, makeProposalFinding, measuredComparisonFromAgentProfileImprovementExperiment, measuredComparisonFromCandidateExperiment, observeCodeAgentSession, paretoPolicy, paretoSignificanceGate, parseAgentTrace, parseCodeAgentJsonl, partitionRunsByAuthoringModel, runAgentProfileImprovementExperiment, runCampaign, runCandidateExperiment, runEval, runImprovementLoop, runReferenceEquivalenceJudge, sealAgentProfileImprovementExperiment, sealAgentProfileImprovementSuite, sealAgentProfileImprovementTask, sealCandidateBenchmarkSuite, sealCandidateBenchmarkTask, sealCandidateExperiment, selfImprove, skillOptOptimizationMethod, summarizeExecution, verifyAgentProfileImprovementExperiment, verifyAgentProfileImprovementExperimentComparison, verifyAgentProfileImprovementSuiteInputs, verifyAgentProfileImprovementTask, verifyCandidateBenchmarkSuite, verifyCandidateBenchmarkSuiteInputs, verifyCandidateBenchmarkTask, verifyCandidateExperiment, verifyCandidateExperimentComparison };
|
|
3733
3727
|
|
|
3734
3728
|
//# sourceMappingURL=index.js.map
|