@kindgi/client 0.1.4-rc.0 → 0.1.4-rc.1

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/dist/index.d.ts CHANGED
@@ -4238,6 +4238,19 @@ declare namespace Schemas {
4238
4238
  version: string;
4239
4239
  conversationId: string;
4240
4240
  };
4241
+ /**
4242
+ * Agents and tools a flow runs at other exact versions than the flow version's pins ("this flow, with `acme.scorer` at 0.4.0"), without publishing a new flow version: a comparison's flow candidate (`versions` on the start body and the run's `comparison`), and the flow runs that replay it (a run's `versions`). Each id must be an agent or tool the flow uses.
4243
+ */
4244
+ export type FlowVersionOverrides = Partial<{
4245
+ /**
4246
+ * Tool id → exact version.
4247
+ */
4248
+ tools: Record<string, string>;
4249
+ /**
4250
+ * Agent id → exact version, for agent nodes with or without a version of their own.
4251
+ */
4252
+ agents: Record<string, string>;
4253
+ }>;
4241
4254
  export type Run = {
4242
4255
  /**
4243
4256
  * RunId.
@@ -4277,6 +4290,7 @@ declare namespace Schemas {
4277
4290
  * Set on a replay run: the eval run that started it.
4278
4291
  */
4279
4292
  evalRunId?: string;
4293
+ versions?: FlowVersionOverrides;
4280
4294
  /**
4281
4295
  * Only in the response to `POST /v1/runs`, when the deployment issues public run tokens: a read-only token for this run (and its descendants) to hand to a browser, for `GET /v1/runs/{runId}/progress` and its stream.
4282
4296
  */
@@ -4394,6 +4408,7 @@ declare namespace Schemas {
4394
4408
  * Set on a replay run: the eval run that started it.
4395
4409
  */
4396
4410
  evalRunId?: string;
4411
+ versions?: FlowVersionOverrides;
4397
4412
  /**
4398
4413
  * Only in the response to `POST /v1/runs`, when the deployment issues public run tokens: a read-only token for this run (and its descendants) to hand to a browser, for `GET /v1/runs/{runId}/progress` and its stream.
4399
4414
  */
@@ -7115,6 +7130,199 @@ declare namespace Schemas {
7115
7130
  reads: "recorded" | "live";
7116
7131
  repetitions: number;
7117
7132
  k: number;
7133
+ versions?: FlowVersionOverrides;
7134
+ };
7135
+ /**
7136
+ * One metric, the recorded runs beside the candidate. `null` where a side had no judged evidence; `n` / `weight` are the candidate's evidence (cases with judged items, and the judgment weight behind them), `baselineN` / `baselineWeight` the recorded side's.
7137
+ */
7138
+ export type ComparisonMetric = {
7139
+ baseline: number | null;
7140
+ candidate: number | null;
7141
+ delta: number | null;
7142
+ n: number;
7143
+ weight: number;
7144
+ baselineN: number;
7145
+ baselineWeight: number;
7146
+ direction: "higher";
7147
+ /**
7148
+ * `weightedPrecisionAtK`: the ranked items it looked at.
7149
+ */
7150
+ k?: number;
7151
+ /**
7152
+ * With more than one repetition: the candidate's max − min across them.
7153
+ */
7154
+ spread?: number;
7155
+ };
7156
+ /**
7157
+ * What ran on the cases: an agent version, or a flow version (with any versions it swapped in).
7158
+ */
7159
+ export type ComparisonCandidate = {
7160
+ kind: "agent";
7161
+ agentId: string;
7162
+ version: string;
7163
+ } | {
7164
+ kind: "flow";
7165
+ flowId: string;
7166
+ version: string;
7167
+ versions?: FlowVersionOverrides;
7168
+ };
7169
+ /**
7170
+ * What a comparison concluded: the candidate beside the recorded runs, the case counts, and the metrics. What a promotion gate reads.
7171
+ */
7172
+ export type JudgedComparisonSummary = {
7173
+ evalRunId: string;
7174
+ status: "completed" | "partial" | "failed";
7175
+ completedAt: string;
7176
+ suite: {
7177
+ id: string;
7178
+ version: string;
7179
+ };
7180
+ candidate: ComparisonCandidate;
7181
+ /**
7182
+ * What the candidate was compared with: `recorded` (the test set's recorded runs, with the versions that served them), or another version.
7183
+ */
7184
+ baseline: {
7185
+ kind: "recorded";
7186
+ versions: Array<{
7187
+ agentId?: string;
7188
+ flowId?: string;
7189
+ version: string;
7190
+ cases: number;
7191
+ }>;
7192
+ } | {
7193
+ kind: "version";
7194
+ agentId: string;
7195
+ version: string;
7196
+ via: "explicit" | "live";
7197
+ liveScope?: Record<string, unknown>;
7198
+ };
7199
+ /**
7200
+ * Where the test set's judgments came from.
7201
+ */
7202
+ scope: Partial<{
7203
+ projectId: string;
7204
+ }>;
7205
+ cases: number;
7206
+ /**
7207
+ * Cases where a read with no recording ran live under `reads: 'recorded'`.
7208
+ */
7209
+ diverged: number;
7210
+ /**
7211
+ * Tool calls refused across the cases (what the candidate would have done).
7212
+ */
7213
+ refusedWrites: number;
7214
+ /**
7215
+ * Cases none of whose repetitions ran.
7216
+ */
7217
+ errors: number;
7218
+ /**
7219
+ * Flow cases that stopped at a write the replay refused: no output to score, so they're left out of the metrics.
7220
+ */
7221
+ stopped: number;
7222
+ reads: "recorded" | "live";
7223
+ sampling: {
7224
+ /**
7225
+ * The models that answered the candidate's replays, and how many replays each.
7226
+ */
7227
+ models: Array<{
7228
+ providerId: string;
7229
+ model: string;
7230
+ runs: number;
7231
+ }>;
7232
+ };
7233
+ repetitions: number;
7234
+ metrics: {
7235
+ weightedYesShare: ComparisonMetric;
7236
+ judgedCoverage: ComparisonMetric;
7237
+ weightedPrecisionAtK: ComparisonMetric;
7238
+ };
7239
+ };
7240
+ /**
7241
+ * One case of a comparison: its replay runs, the scores, the items kept, dropped and new, the tool calls, and why it didn't run when it didn't.
7242
+ */
7243
+ export type ComparisonCaseResult = {
7244
+ caseId: string;
7245
+ /**
7246
+ * The candidate's replay runs, one per repetition.
7247
+ */
7248
+ runIds: Array<string>;
7249
+ /**
7250
+ * An output's score: Σ yesWeight and Σ totalWeight over its judged items, and over those among the first `k` ranked items.
7251
+ */
7252
+ baseline: {
7253
+ yesWeight: number;
7254
+ totalWeight: number;
7255
+ items: number;
7256
+ judgedItems: number;
7257
+ topK: {
7258
+ yesWeight: number;
7259
+ totalWeight: number;
7260
+ };
7261
+ };
7262
+ /**
7263
+ * One per repetition that ran.
7264
+ */
7265
+ candidate: Array<{
7266
+ yesWeight: number;
7267
+ totalWeight: number;
7268
+ items: number;
7269
+ judgedItems: number;
7270
+ topK: {
7271
+ yesWeight: number;
7272
+ totalWeight: number;
7273
+ };
7274
+ }>;
7275
+ /**
7276
+ * The first repetition's items against the judged ones.
7277
+ */
7278
+ changes?: {
7279
+ kept: Array<{
7280
+ key: string;
7281
+ rankBefore?: number;
7282
+ rank?: number;
7283
+ }>;
7284
+ dropped: Array<{
7285
+ key: string;
7286
+ rankBefore?: number;
7287
+ }>;
7288
+ new: Array<{
7289
+ key: string;
7290
+ pointer: string;
7291
+ rank?: number;
7292
+ }>;
7293
+ };
7294
+ /**
7295
+ * The first repetition's tool calls, and what happened to each.
7296
+ */
7297
+ tools?: Array<{
7298
+ step: number;
7299
+ callId: string;
7300
+ toolId: string;
7301
+ toolVersion: string;
7302
+ arguments: unknown;
7303
+ source: "live" | "recorded" | "refused";
7304
+ reason?: string;
7305
+ }>;
7306
+ diverged: boolean;
7307
+ refusedWrites: number;
7308
+ noContext: boolean;
7309
+ approvalSkipped: boolean;
7310
+ error?: string;
7311
+ /**
7312
+ * Set when the replay stopped at a refused write: what it would have done.
7313
+ */
7314
+ stopped?: {
7315
+ toolId: string;
7316
+ arguments: unknown;
7317
+ reason?: string;
7318
+ };
7319
+ };
7320
+ /**
7321
+ * A comparison's `result` (a `judged` eval run's): the summary and each case. `EvalRun.result` stays an open object, since each kind has its own; the clients read it as this (TS `comparisonOf(run)`, Python `comparison_of(run)`).
7322
+ */
7323
+ export type JudgedComparisonResult = {
7324
+ summary: JudgedComparisonSummary;
7325
+ perCase: Array<ComparisonCaseResult>;
7118
7326
  };
7119
7327
  export type EvalRun = {
7120
7328
  runId: string;
@@ -7129,7 +7337,7 @@ declare namespace Schemas {
7129
7337
  startedAt: string;
7130
7338
  completedAt?: string;
7131
7339
  /**
7132
- * Kind-specific opaque JSON. For `accuracy`, contains `{ passCount, totalCount, meanScore, perCase[] }`. For `judged` (a comparison), `{ summary, perCase[] }`: the summary has the baseline (the versions behind the recorded runs) and the candidate (`{ kind: "agent", agentId, version }` or `{ kind: "flow", flowId, version }`), the case counts (`cases`, `diverged`, `refusedWrites`, `errors`, and `stopped`: flow cases that stopped at a write the replay refused, left out of the metrics), the models that answered, and `metrics` (`weightedYesShare`, `judgedCoverage`, `weightedPrecisionAtK`, each `{ baseline, candidate, delta, n, weight, baselineN, baselineWeight, direction, k?, spread? }`); each case has its replay runs, the scores, the items kept, dropped and new, the tool calls with what happened to each, and `stopped` (what it would have done) when it stopped. Other kinds define their own shapes as their dispatchers ship.
7340
+ * Kind-specific opaque JSON. For `accuracy`, contains `{ passCount, totalCount, meanScore, perCase[] }`. For `judged` (a comparison), `{ summary, perCase[] }`: the summary has the baseline (the versions behind the recorded runs) and the candidate (`{ kind: "agent", agentId, version }` or `{ kind: "flow", flowId, version, versions? }`), the case counts (`cases`, `diverged`, `refusedWrites`, `errors`, and `stopped`: flow cases that stopped at a write the replay refused, left out of the metrics), the models that answered, and `metrics` (`weightedYesShare`, `judgedCoverage`, `weightedPrecisionAtK`, each `{ baseline, candidate, delta, n, weight, baselineN, baselineWeight, direction, k?, spread? }`); each case has its replay runs, the scores, the items kept, dropped and new, the tool calls with what happened to each, and `stopped` (what it would have done) when it stopped. Other kinds define their own shapes as their dispatchers ship.
7133
7341
  */
7134
7342
  result?: Record<string, unknown>;
7135
7343
  error?: string;
@@ -7142,7 +7350,7 @@ declare namespace Schemas {
7142
7350
  hasMore: boolean;
7143
7351
  };
7144
7352
  /**
7145
- * Exactly one of `agentRef` or `flowRef` MUST be supplied. `dryRun: true` returns a plan preview without invoking the subject. For a `judged` suite (a test set), the run is a comparison: `agentRef` or `flowRef` with its `version` is the candidate, replayed on each case without doing anything the past run didn't (a flow stops at a write the replay refuses); `baseline` (default `'recorded'`), `reads` (default `recorded`), `repetitions` (default 1) and `k` (default 10) set how.
7353
+ * Exactly one of `agentRef` or `flowRef` MUST be supplied. `dryRun: true` returns a plan preview without invoking the subject. For a `judged` suite (a test set), the run is a comparison: `agentRef` or `flowRef` with its `version` is the candidate, replayed on each case without doing anything the past run didn't (a flow stops at a write the replay refuses); `baseline` (default `'recorded'`), `reads` (default `recorded`), `repetitions` (default 1) and `k` (default 10) set how. With `flowRef`, `versions` runs the flow with some of its agents or tools at other versions; an id the flow doesn't use, or a version that isn't published, is refused (`400 validation-failed`, each under `details.issues`).
7146
7354
  */
7147
7355
  export type StartEvalRunBody = {
7148
7356
  /**
@@ -7157,6 +7365,7 @@ declare namespace Schemas {
7157
7365
  reads?: "recorded" | "live";
7158
7366
  repetitions?: number;
7159
7367
  k?: number;
7368
+ versions?: FlowVersionOverrides;
7160
7369
  };
7161
7370
  export type StartEvalRunResult = {
7162
7371
  runId: string;
@@ -9019,6 +9228,383 @@ declare const EvalRunStatus: z.ZodEnum<{
9019
9228
  completed: "completed";
9020
9229
  cancelled: "cancelled";
9021
9230
  }>;
9231
+ export type ComparisonMetric = Schemas.ComparisonMetric;
9232
+ export declare const ComparisonMetric: z.ZodObject<{
9233
+ baseline: z.ZodNullable<z.ZodNumber>;
9234
+ candidate: z.ZodNullable<z.ZodNumber>;
9235
+ delta: z.ZodNullable<z.ZodNumber>;
9236
+ n: z.ZodNumber;
9237
+ weight: z.ZodNumber;
9238
+ baselineN: z.ZodNumber;
9239
+ baselineWeight: z.ZodNumber;
9240
+ direction: z.ZodLiteral<"higher">;
9241
+ k: z.ZodOptional<z.ZodNumber>;
9242
+ spread: z.ZodOptional<z.ZodNumber>;
9243
+ }, z.core.$strict>;
9244
+ export type ComparisonCandidate = Schemas.ComparisonCandidate;
9245
+ export declare const ComparisonCandidate: z.ZodUnion<readonly [
9246
+ z.ZodObject<{
9247
+ kind: z.ZodLiteral<"agent">;
9248
+ agentId: z.ZodString;
9249
+ version: z.ZodString;
9250
+ }, z.core.$strict>,
9251
+ z.ZodObject<{
9252
+ kind: z.ZodLiteral<"flow">;
9253
+ flowId: z.ZodString;
9254
+ version: z.ZodString;
9255
+ versions: z.ZodOptional<z.ZodObject<{
9256
+ tools: z.ZodOptional<z.ZodRecord<z.ZodString, z.ZodString>>;
9257
+ agents: z.ZodOptional<z.ZodRecord<z.ZodString, z.ZodString>>;
9258
+ }, z.core.$strict>>;
9259
+ }, z.core.$strict>
9260
+ ]>;
9261
+ export type JudgedComparisonSummary = Schemas.JudgedComparisonSummary;
9262
+ export declare const JudgedComparisonSummary: z.ZodObject<{
9263
+ evalRunId: z.ZodString;
9264
+ status: z.ZodEnum<{
9265
+ failed: "failed";
9266
+ completed: "completed";
9267
+ partial: "partial";
9268
+ }>;
9269
+ completedAt: z.ZodISODateTime;
9270
+ suite: z.ZodObject<{
9271
+ id: z.ZodString;
9272
+ version: z.ZodString;
9273
+ }, z.core.$strict>;
9274
+ candidate: z.ZodUnion<readonly [
9275
+ z.ZodObject<{
9276
+ kind: z.ZodLiteral<"agent">;
9277
+ agentId: z.ZodString;
9278
+ version: z.ZodString;
9279
+ }, z.core.$strict>,
9280
+ z.ZodObject<{
9281
+ kind: z.ZodLiteral<"flow">;
9282
+ flowId: z.ZodString;
9283
+ version: z.ZodString;
9284
+ versions: z.ZodOptional<z.ZodObject<{
9285
+ tools: z.ZodOptional<z.ZodRecord<z.ZodString, z.ZodString>>;
9286
+ agents: z.ZodOptional<z.ZodRecord<z.ZodString, z.ZodString>>;
9287
+ }, z.core.$strict>>;
9288
+ }, z.core.$strict>
9289
+ ]>;
9290
+ baseline: z.ZodUnion<readonly [
9291
+ z.ZodObject<{
9292
+ kind: z.ZodLiteral<"recorded">;
9293
+ versions: z.ZodArray<z.ZodObject<{
9294
+ agentId: z.ZodOptional<z.ZodString>;
9295
+ flowId: z.ZodOptional<z.ZodString>;
9296
+ version: z.ZodString;
9297
+ cases: z.ZodNumber;
9298
+ }, z.core.$strict>>;
9299
+ }, z.core.$strict>,
9300
+ z.ZodObject<{
9301
+ kind: z.ZodLiteral<"version">;
9302
+ agentId: z.ZodString;
9303
+ version: z.ZodString;
9304
+ via: z.ZodEnum<{
9305
+ live: "live";
9306
+ explicit: "explicit";
9307
+ }>;
9308
+ liveScope: z.ZodOptional<z.ZodRecord<z.ZodString, z.ZodUnknown>>;
9309
+ }, z.core.$strict>
9310
+ ]>;
9311
+ scope: z.ZodObject<{
9312
+ projectId: z.ZodOptional<z.ZodString>;
9313
+ }, z.core.$strict>;
9314
+ cases: z.ZodNumber;
9315
+ diverged: z.ZodNumber;
9316
+ refusedWrites: z.ZodNumber;
9317
+ errors: z.ZodNumber;
9318
+ stopped: z.ZodNumber;
9319
+ reads: z.ZodEnum<{
9320
+ recorded: "recorded";
9321
+ live: "live";
9322
+ }>;
9323
+ sampling: z.ZodObject<{
9324
+ models: z.ZodArray<z.ZodObject<{
9325
+ providerId: z.ZodString;
9326
+ model: z.ZodString;
9327
+ runs: z.ZodNumber;
9328
+ }, z.core.$strict>>;
9329
+ }, z.core.$strict>;
9330
+ repetitions: z.ZodNumber;
9331
+ metrics: z.ZodObject<{
9332
+ weightedYesShare: z.ZodObject<{
9333
+ baseline: z.ZodNullable<z.ZodNumber>;
9334
+ candidate: z.ZodNullable<z.ZodNumber>;
9335
+ delta: z.ZodNullable<z.ZodNumber>;
9336
+ n: z.ZodNumber;
9337
+ weight: z.ZodNumber;
9338
+ baselineN: z.ZodNumber;
9339
+ baselineWeight: z.ZodNumber;
9340
+ direction: z.ZodLiteral<"higher">;
9341
+ k: z.ZodOptional<z.ZodNumber>;
9342
+ spread: z.ZodOptional<z.ZodNumber>;
9343
+ }, z.core.$strict>;
9344
+ judgedCoverage: z.ZodObject<{
9345
+ baseline: z.ZodNullable<z.ZodNumber>;
9346
+ candidate: z.ZodNullable<z.ZodNumber>;
9347
+ delta: z.ZodNullable<z.ZodNumber>;
9348
+ n: z.ZodNumber;
9349
+ weight: z.ZodNumber;
9350
+ baselineN: z.ZodNumber;
9351
+ baselineWeight: z.ZodNumber;
9352
+ direction: z.ZodLiteral<"higher">;
9353
+ k: z.ZodOptional<z.ZodNumber>;
9354
+ spread: z.ZodOptional<z.ZodNumber>;
9355
+ }, z.core.$strict>;
9356
+ weightedPrecisionAtK: z.ZodObject<{
9357
+ baseline: z.ZodNullable<z.ZodNumber>;
9358
+ candidate: z.ZodNullable<z.ZodNumber>;
9359
+ delta: z.ZodNullable<z.ZodNumber>;
9360
+ n: z.ZodNumber;
9361
+ weight: z.ZodNumber;
9362
+ baselineN: z.ZodNumber;
9363
+ baselineWeight: z.ZodNumber;
9364
+ direction: z.ZodLiteral<"higher">;
9365
+ k: z.ZodOptional<z.ZodNumber>;
9366
+ spread: z.ZodOptional<z.ZodNumber>;
9367
+ }, z.core.$strict>;
9368
+ }, z.core.$strict>;
9369
+ }, z.core.$strict>;
9370
+ export type ComparisonCaseResult = Schemas.ComparisonCaseResult;
9371
+ export declare const ComparisonCaseResult: z.ZodObject<{
9372
+ caseId: z.ZodString;
9373
+ runIds: z.ZodArray<z.ZodString>;
9374
+ baseline: z.ZodObject<{
9375
+ yesWeight: z.ZodNumber;
9376
+ totalWeight: z.ZodNumber;
9377
+ items: z.ZodNumber;
9378
+ judgedItems: z.ZodNumber;
9379
+ topK: z.ZodObject<{
9380
+ yesWeight: z.ZodNumber;
9381
+ totalWeight: z.ZodNumber;
9382
+ }, z.core.$strict>;
9383
+ }, z.core.$strict>;
9384
+ candidate: z.ZodArray<z.ZodObject<{
9385
+ yesWeight: z.ZodNumber;
9386
+ totalWeight: z.ZodNumber;
9387
+ items: z.ZodNumber;
9388
+ judgedItems: z.ZodNumber;
9389
+ topK: z.ZodObject<{
9390
+ yesWeight: z.ZodNumber;
9391
+ totalWeight: z.ZodNumber;
9392
+ }, z.core.$strict>;
9393
+ }, z.core.$strict>>;
9394
+ changes: z.ZodOptional<z.ZodObject<{
9395
+ kept: z.ZodArray<z.ZodObject<{
9396
+ key: z.ZodString;
9397
+ rankBefore: z.ZodOptional<z.ZodNumber>;
9398
+ rank: z.ZodOptional<z.ZodNumber>;
9399
+ }, z.core.$strict>>;
9400
+ dropped: z.ZodArray<z.ZodObject<{
9401
+ key: z.ZodString;
9402
+ rankBefore: z.ZodOptional<z.ZodNumber>;
9403
+ }, z.core.$strict>>;
9404
+ new: z.ZodArray<z.ZodObject<{
9405
+ key: z.ZodString;
9406
+ pointer: z.ZodString;
9407
+ rank: z.ZodOptional<z.ZodNumber>;
9408
+ }, z.core.$strict>>;
9409
+ }, z.core.$strict>>;
9410
+ tools: z.ZodOptional<z.ZodArray<z.ZodObject<{
9411
+ step: z.ZodNumber;
9412
+ callId: z.ZodString;
9413
+ toolId: z.ZodString;
9414
+ toolVersion: z.ZodString;
9415
+ arguments: z.ZodUnknown;
9416
+ source: z.ZodEnum<{
9417
+ recorded: "recorded";
9418
+ live: "live";
9419
+ refused: "refused";
9420
+ }>;
9421
+ reason: z.ZodOptional<z.ZodString>;
9422
+ }, z.core.$strict>>>;
9423
+ diverged: z.ZodBoolean;
9424
+ refusedWrites: z.ZodNumber;
9425
+ noContext: z.ZodBoolean;
9426
+ approvalSkipped: z.ZodBoolean;
9427
+ error: z.ZodOptional<z.ZodString>;
9428
+ stopped: z.ZodOptional<z.ZodObject<{
9429
+ toolId: z.ZodString;
9430
+ arguments: z.ZodUnknown;
9431
+ reason: z.ZodOptional<z.ZodString>;
9432
+ }, z.core.$strict>>;
9433
+ }, z.core.$strict>;
9434
+ export type JudgedComparisonResult = Schemas.JudgedComparisonResult;
9435
+ export declare const JudgedComparisonResult: z.ZodObject<{
9436
+ summary: z.ZodObject<{
9437
+ evalRunId: z.ZodString;
9438
+ status: z.ZodEnum<{
9439
+ failed: "failed";
9440
+ completed: "completed";
9441
+ partial: "partial";
9442
+ }>;
9443
+ completedAt: z.ZodISODateTime;
9444
+ suite: z.ZodObject<{
9445
+ id: z.ZodString;
9446
+ version: z.ZodString;
9447
+ }, z.core.$strict>;
9448
+ candidate: z.ZodUnion<readonly [
9449
+ z.ZodObject<{
9450
+ kind: z.ZodLiteral<"agent">;
9451
+ agentId: z.ZodString;
9452
+ version: z.ZodString;
9453
+ }, z.core.$strict>,
9454
+ z.ZodObject<{
9455
+ kind: z.ZodLiteral<"flow">;
9456
+ flowId: z.ZodString;
9457
+ version: z.ZodString;
9458
+ versions: z.ZodOptional<z.ZodObject<{
9459
+ tools: z.ZodOptional<z.ZodRecord<z.ZodString, z.ZodString>>;
9460
+ agents: z.ZodOptional<z.ZodRecord<z.ZodString, z.ZodString>>;
9461
+ }, z.core.$strict>>;
9462
+ }, z.core.$strict>
9463
+ ]>;
9464
+ baseline: z.ZodUnion<readonly [
9465
+ z.ZodObject<{
9466
+ kind: z.ZodLiteral<"recorded">;
9467
+ versions: z.ZodArray<z.ZodObject<{
9468
+ agentId: z.ZodOptional<z.ZodString>;
9469
+ flowId: z.ZodOptional<z.ZodString>;
9470
+ version: z.ZodString;
9471
+ cases: z.ZodNumber;
9472
+ }, z.core.$strict>>;
9473
+ }, z.core.$strict>,
9474
+ z.ZodObject<{
9475
+ kind: z.ZodLiteral<"version">;
9476
+ agentId: z.ZodString;
9477
+ version: z.ZodString;
9478
+ via: z.ZodEnum<{
9479
+ live: "live";
9480
+ explicit: "explicit";
9481
+ }>;
9482
+ liveScope: z.ZodOptional<z.ZodRecord<z.ZodString, z.ZodUnknown>>;
9483
+ }, z.core.$strict>
9484
+ ]>;
9485
+ scope: z.ZodObject<{
9486
+ projectId: z.ZodOptional<z.ZodString>;
9487
+ }, z.core.$strict>;
9488
+ cases: z.ZodNumber;
9489
+ diverged: z.ZodNumber;
9490
+ refusedWrites: z.ZodNumber;
9491
+ errors: z.ZodNumber;
9492
+ stopped: z.ZodNumber;
9493
+ reads: z.ZodEnum<{
9494
+ recorded: "recorded";
9495
+ live: "live";
9496
+ }>;
9497
+ sampling: z.ZodObject<{
9498
+ models: z.ZodArray<z.ZodObject<{
9499
+ providerId: z.ZodString;
9500
+ model: z.ZodString;
9501
+ runs: z.ZodNumber;
9502
+ }, z.core.$strict>>;
9503
+ }, z.core.$strict>;
9504
+ repetitions: z.ZodNumber;
9505
+ metrics: z.ZodObject<{
9506
+ weightedYesShare: z.ZodObject<{
9507
+ baseline: z.ZodNullable<z.ZodNumber>;
9508
+ candidate: z.ZodNullable<z.ZodNumber>;
9509
+ delta: z.ZodNullable<z.ZodNumber>;
9510
+ n: z.ZodNumber;
9511
+ weight: z.ZodNumber;
9512
+ baselineN: z.ZodNumber;
9513
+ baselineWeight: z.ZodNumber;
9514
+ direction: z.ZodLiteral<"higher">;
9515
+ k: z.ZodOptional<z.ZodNumber>;
9516
+ spread: z.ZodOptional<z.ZodNumber>;
9517
+ }, z.core.$strict>;
9518
+ judgedCoverage: z.ZodObject<{
9519
+ baseline: z.ZodNullable<z.ZodNumber>;
9520
+ candidate: z.ZodNullable<z.ZodNumber>;
9521
+ delta: z.ZodNullable<z.ZodNumber>;
9522
+ n: z.ZodNumber;
9523
+ weight: z.ZodNumber;
9524
+ baselineN: z.ZodNumber;
9525
+ baselineWeight: z.ZodNumber;
9526
+ direction: z.ZodLiteral<"higher">;
9527
+ k: z.ZodOptional<z.ZodNumber>;
9528
+ spread: z.ZodOptional<z.ZodNumber>;
9529
+ }, z.core.$strict>;
9530
+ weightedPrecisionAtK: z.ZodObject<{
9531
+ baseline: z.ZodNullable<z.ZodNumber>;
9532
+ candidate: z.ZodNullable<z.ZodNumber>;
9533
+ delta: z.ZodNullable<z.ZodNumber>;
9534
+ n: z.ZodNumber;
9535
+ weight: z.ZodNumber;
9536
+ baselineN: z.ZodNumber;
9537
+ baselineWeight: z.ZodNumber;
9538
+ direction: z.ZodLiteral<"higher">;
9539
+ k: z.ZodOptional<z.ZodNumber>;
9540
+ spread: z.ZodOptional<z.ZodNumber>;
9541
+ }, z.core.$strict>;
9542
+ }, z.core.$strict>;
9543
+ }, z.core.$strict>;
9544
+ perCase: z.ZodArray<z.ZodObject<{
9545
+ caseId: z.ZodString;
9546
+ runIds: z.ZodArray<z.ZodString>;
9547
+ baseline: z.ZodObject<{
9548
+ yesWeight: z.ZodNumber;
9549
+ totalWeight: z.ZodNumber;
9550
+ items: z.ZodNumber;
9551
+ judgedItems: z.ZodNumber;
9552
+ topK: z.ZodObject<{
9553
+ yesWeight: z.ZodNumber;
9554
+ totalWeight: z.ZodNumber;
9555
+ }, z.core.$strict>;
9556
+ }, z.core.$strict>;
9557
+ candidate: z.ZodArray<z.ZodObject<{
9558
+ yesWeight: z.ZodNumber;
9559
+ totalWeight: z.ZodNumber;
9560
+ items: z.ZodNumber;
9561
+ judgedItems: z.ZodNumber;
9562
+ topK: z.ZodObject<{
9563
+ yesWeight: z.ZodNumber;
9564
+ totalWeight: z.ZodNumber;
9565
+ }, z.core.$strict>;
9566
+ }, z.core.$strict>>;
9567
+ changes: z.ZodOptional<z.ZodObject<{
9568
+ kept: z.ZodArray<z.ZodObject<{
9569
+ key: z.ZodString;
9570
+ rankBefore: z.ZodOptional<z.ZodNumber>;
9571
+ rank: z.ZodOptional<z.ZodNumber>;
9572
+ }, z.core.$strict>>;
9573
+ dropped: z.ZodArray<z.ZodObject<{
9574
+ key: z.ZodString;
9575
+ rankBefore: z.ZodOptional<z.ZodNumber>;
9576
+ }, z.core.$strict>>;
9577
+ new: z.ZodArray<z.ZodObject<{
9578
+ key: z.ZodString;
9579
+ pointer: z.ZodString;
9580
+ rank: z.ZodOptional<z.ZodNumber>;
9581
+ }, z.core.$strict>>;
9582
+ }, z.core.$strict>>;
9583
+ tools: z.ZodOptional<z.ZodArray<z.ZodObject<{
9584
+ step: z.ZodNumber;
9585
+ callId: z.ZodString;
9586
+ toolId: z.ZodString;
9587
+ toolVersion: z.ZodString;
9588
+ arguments: z.ZodUnknown;
9589
+ source: z.ZodEnum<{
9590
+ recorded: "recorded";
9591
+ live: "live";
9592
+ refused: "refused";
9593
+ }>;
9594
+ reason: z.ZodOptional<z.ZodString>;
9595
+ }, z.core.$strict>>>;
9596
+ diverged: z.ZodBoolean;
9597
+ refusedWrites: z.ZodNumber;
9598
+ noContext: z.ZodBoolean;
9599
+ approvalSkipped: z.ZodBoolean;
9600
+ error: z.ZodOptional<z.ZodString>;
9601
+ stopped: z.ZodOptional<z.ZodObject<{
9602
+ toolId: z.ZodString;
9603
+ arguments: z.ZodUnknown;
9604
+ reason: z.ZodOptional<z.ZodString>;
9605
+ }, z.core.$strict>>;
9606
+ }, z.core.$strict>>;
9607
+ }, z.core.$strict>;
9022
9608
  export type EvalRun = Schemas.EvalRun;
9023
9609
  declare const EvalRun: z.ZodObject<{
9024
9610
  runId: z.ZodUUID;
@@ -9075,6 +9661,10 @@ declare const EvalRun: z.ZodObject<{
9075
9661
  }>;
9076
9662
  repetitions: z.ZodNumber;
9077
9663
  k: z.ZodNumber;
9664
+ versions: z.ZodOptional<z.ZodObject<{
9665
+ tools: z.ZodOptional<z.ZodRecord<z.ZodString, z.ZodString>>;
9666
+ agents: z.ZodOptional<z.ZodRecord<z.ZodString, z.ZodString>>;
9667
+ }, z.core.$strict>>;
9078
9668
  }, z.core.$strict>>;
9079
9669
  }, z.core.$strict>;
9080
9670
  export type EvalRunCollectionPage = Schemas.EvalRunCollectionPage;
@@ -9134,6 +9724,10 @@ declare const EvalRunCollectionPage: z.ZodObject<{
9134
9724
  }>;
9135
9725
  repetitions: z.ZodNumber;
9136
9726
  k: z.ZodNumber;
9727
+ versions: z.ZodOptional<z.ZodObject<{
9728
+ tools: z.ZodOptional<z.ZodRecord<z.ZodString, z.ZodString>>;
9729
+ agents: z.ZodOptional<z.ZodRecord<z.ZodString, z.ZodString>>;
9730
+ }, z.core.$strict>>;
9137
9731
  }, z.core.$strict>>;
9138
9732
  }, z.core.$strict>>;
9139
9733
  nextCursor: z.ZodOptional<z.ZodString>;
@@ -9171,6 +9765,10 @@ declare const StartEvalRunBody: z.ZodObject<{
9171
9765
  }>>;
9172
9766
  repetitions: z.ZodOptional<z.ZodNumber>;
9173
9767
  k: z.ZodOptional<z.ZodNumber>;
9768
+ versions: z.ZodOptional<z.ZodObject<{
9769
+ tools: z.ZodOptional<z.ZodRecord<z.ZodString, z.ZodString>>;
9770
+ agents: z.ZodOptional<z.ZodRecord<z.ZodString, z.ZodString>>;
9771
+ }, z.core.$strict>>;
9174
9772
  }, z.core.$strict>;
9175
9773
  export type StartEvalRunResult = Schemas.StartEvalRunResult;
9176
9774
  declare const StartEvalRunResult: z.ZodObject<{
@@ -11338,6 +11936,13 @@ export type EvalRunPage = EvalRunCollectionPage;
11338
11936
  */
11339
11937
  export type StartEvalRunInput = Omit<StartEvalRunBody, "projectId">;
11340
11938
  export type StartEvalRunOutcome = StartEvalRunResult;
11939
+ /**
11940
+ * A comparison's result, typed from the OpenAPI `JudgedComparisonResult`:
11941
+ * its summary and each case. `run` is a `judged` eval run (a test set
11942
+ * compared with a version); `undefined` for another kind of eval run, a
11943
+ * dry run, or one that hasn't finished.
11944
+ */
11945
+ export declare function comparisonOf(run: EvalRunRecord): JudgedComparisonResult | undefined;
11341
11946
  export interface StartEvalRunOptions {
11342
11947
  /** Project the eval run belongs to. The route requires it. */
11343
11948
  readonly projectId: string;