@kindgi/client 0.1.4-rc.0 → 0.1.4-rc.1
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/dist/index.cjs +6 -0
- package/dist/index.d.cts +607 -2
- package/dist/index.d.ts +607 -2
- package/dist/index.js +6 -1
- package/package.json +7 -7
package/dist/index.d.ts
CHANGED
|
@@ -4238,6 +4238,19 @@ declare namespace Schemas {
|
|
|
4238
4238
|
version: string;
|
|
4239
4239
|
conversationId: string;
|
|
4240
4240
|
};
|
|
4241
|
+
/**
|
|
4242
|
+
* Agents and tools a flow runs at other exact versions than the flow version's pins ("this flow, with `acme.scorer` at 0.4.0"), without publishing a new flow version: a comparison's flow candidate (`versions` on the start body and the run's `comparison`), and the flow runs that replay it (a run's `versions`). Each id must be an agent or tool the flow uses.
|
|
4243
|
+
*/
|
|
4244
|
+
export type FlowVersionOverrides = Partial<{
|
|
4245
|
+
/**
|
|
4246
|
+
* Tool id → exact version.
|
|
4247
|
+
*/
|
|
4248
|
+
tools: Record<string, string>;
|
|
4249
|
+
/**
|
|
4250
|
+
* Agent id → exact version, for agent nodes with or without a version of their own.
|
|
4251
|
+
*/
|
|
4252
|
+
agents: Record<string, string>;
|
|
4253
|
+
}>;
|
|
4241
4254
|
export type Run = {
|
|
4242
4255
|
/**
|
|
4243
4256
|
* RunId.
|
|
@@ -4277,6 +4290,7 @@ declare namespace Schemas {
|
|
|
4277
4290
|
* Set on a replay run: the eval run that started it.
|
|
4278
4291
|
*/
|
|
4279
4292
|
evalRunId?: string;
|
|
4293
|
+
versions?: FlowVersionOverrides;
|
|
4280
4294
|
/**
|
|
4281
4295
|
* Only in the response to `POST /v1/runs`, when the deployment issues public run tokens: a read-only token for this run (and its descendants) to hand to a browser, for `GET /v1/runs/{runId}/progress` and its stream.
|
|
4282
4296
|
*/
|
|
@@ -4394,6 +4408,7 @@ declare namespace Schemas {
|
|
|
4394
4408
|
* Set on a replay run: the eval run that started it.
|
|
4395
4409
|
*/
|
|
4396
4410
|
evalRunId?: string;
|
|
4411
|
+
versions?: FlowVersionOverrides;
|
|
4397
4412
|
/**
|
|
4398
4413
|
* Only in the response to `POST /v1/runs`, when the deployment issues public run tokens: a read-only token for this run (and its descendants) to hand to a browser, for `GET /v1/runs/{runId}/progress` and its stream.
|
|
4399
4414
|
*/
|
|
@@ -7115,6 +7130,199 @@ declare namespace Schemas {
|
|
|
7115
7130
|
reads: "recorded" | "live";
|
|
7116
7131
|
repetitions: number;
|
|
7117
7132
|
k: number;
|
|
7133
|
+
versions?: FlowVersionOverrides;
|
|
7134
|
+
};
|
|
7135
|
+
/**
|
|
7136
|
+
* One metric, the recorded runs beside the candidate. `null` where a side had no judged evidence; `n` / `weight` are the candidate's evidence (cases with judged items, and the judgment weight behind them), `baselineN` / `baselineWeight` the recorded side's.
|
|
7137
|
+
*/
|
|
7138
|
+
export type ComparisonMetric = {
|
|
7139
|
+
baseline: number | null;
|
|
7140
|
+
candidate: number | null;
|
|
7141
|
+
delta: number | null;
|
|
7142
|
+
n: number;
|
|
7143
|
+
weight: number;
|
|
7144
|
+
baselineN: number;
|
|
7145
|
+
baselineWeight: number;
|
|
7146
|
+
direction: "higher";
|
|
7147
|
+
/**
|
|
7148
|
+
* `weightedPrecisionAtK`: the ranked items it looked at.
|
|
7149
|
+
*/
|
|
7150
|
+
k?: number;
|
|
7151
|
+
/**
|
|
7152
|
+
* With more than one repetition: the candidate's max − min across them.
|
|
7153
|
+
*/
|
|
7154
|
+
spread?: number;
|
|
7155
|
+
};
|
|
7156
|
+
/**
|
|
7157
|
+
* What ran on the cases: an agent version, or a flow version (with any versions it swapped in).
|
|
7158
|
+
*/
|
|
7159
|
+
export type ComparisonCandidate = {
|
|
7160
|
+
kind: "agent";
|
|
7161
|
+
agentId: string;
|
|
7162
|
+
version: string;
|
|
7163
|
+
} | {
|
|
7164
|
+
kind: "flow";
|
|
7165
|
+
flowId: string;
|
|
7166
|
+
version: string;
|
|
7167
|
+
versions?: FlowVersionOverrides;
|
|
7168
|
+
};
|
|
7169
|
+
/**
|
|
7170
|
+
* What a comparison concluded: the candidate beside the recorded runs, the case counts, and the metrics. What a promotion gate reads.
|
|
7171
|
+
*/
|
|
7172
|
+
export type JudgedComparisonSummary = {
|
|
7173
|
+
evalRunId: string;
|
|
7174
|
+
status: "completed" | "partial" | "failed";
|
|
7175
|
+
completedAt: string;
|
|
7176
|
+
suite: {
|
|
7177
|
+
id: string;
|
|
7178
|
+
version: string;
|
|
7179
|
+
};
|
|
7180
|
+
candidate: ComparisonCandidate;
|
|
7181
|
+
/**
|
|
7182
|
+
* What the candidate was compared with: `recorded` (the test set's recorded runs, with the versions that served them), or another version.
|
|
7183
|
+
*/
|
|
7184
|
+
baseline: {
|
|
7185
|
+
kind: "recorded";
|
|
7186
|
+
versions: Array<{
|
|
7187
|
+
agentId?: string;
|
|
7188
|
+
flowId?: string;
|
|
7189
|
+
version: string;
|
|
7190
|
+
cases: number;
|
|
7191
|
+
}>;
|
|
7192
|
+
} | {
|
|
7193
|
+
kind: "version";
|
|
7194
|
+
agentId: string;
|
|
7195
|
+
version: string;
|
|
7196
|
+
via: "explicit" | "live";
|
|
7197
|
+
liveScope?: Record<string, unknown>;
|
|
7198
|
+
};
|
|
7199
|
+
/**
|
|
7200
|
+
* Where the test set's judgments came from.
|
|
7201
|
+
*/
|
|
7202
|
+
scope: Partial<{
|
|
7203
|
+
projectId: string;
|
|
7204
|
+
}>;
|
|
7205
|
+
cases: number;
|
|
7206
|
+
/**
|
|
7207
|
+
* Cases where a read with no recording ran live under `reads: 'recorded'`.
|
|
7208
|
+
*/
|
|
7209
|
+
diverged: number;
|
|
7210
|
+
/**
|
|
7211
|
+
* Tool calls refused across the cases (what the candidate would have done).
|
|
7212
|
+
*/
|
|
7213
|
+
refusedWrites: number;
|
|
7214
|
+
/**
|
|
7215
|
+
* Cases none of whose repetitions ran.
|
|
7216
|
+
*/
|
|
7217
|
+
errors: number;
|
|
7218
|
+
/**
|
|
7219
|
+
* Flow cases that stopped at a write the replay refused: no output to score, so they're left out of the metrics.
|
|
7220
|
+
*/
|
|
7221
|
+
stopped: number;
|
|
7222
|
+
reads: "recorded" | "live";
|
|
7223
|
+
sampling: {
|
|
7224
|
+
/**
|
|
7225
|
+
* The models that answered the candidate's replays, and how many replays each.
|
|
7226
|
+
*/
|
|
7227
|
+
models: Array<{
|
|
7228
|
+
providerId: string;
|
|
7229
|
+
model: string;
|
|
7230
|
+
runs: number;
|
|
7231
|
+
}>;
|
|
7232
|
+
};
|
|
7233
|
+
repetitions: number;
|
|
7234
|
+
metrics: {
|
|
7235
|
+
weightedYesShare: ComparisonMetric;
|
|
7236
|
+
judgedCoverage: ComparisonMetric;
|
|
7237
|
+
weightedPrecisionAtK: ComparisonMetric;
|
|
7238
|
+
};
|
|
7239
|
+
};
|
|
7240
|
+
/**
|
|
7241
|
+
* One case of a comparison: its replay runs, the scores, the items kept, dropped and new, the tool calls, and why it didn't run when it didn't.
|
|
7242
|
+
*/
|
|
7243
|
+
export type ComparisonCaseResult = {
|
|
7244
|
+
caseId: string;
|
|
7245
|
+
/**
|
|
7246
|
+
* The candidate's replay runs, one per repetition.
|
|
7247
|
+
*/
|
|
7248
|
+
runIds: Array<string>;
|
|
7249
|
+
/**
|
|
7250
|
+
* An output's score: Σ yesWeight and Σ totalWeight over its judged items, and over those among the first `k` ranked items.
|
|
7251
|
+
*/
|
|
7252
|
+
baseline: {
|
|
7253
|
+
yesWeight: number;
|
|
7254
|
+
totalWeight: number;
|
|
7255
|
+
items: number;
|
|
7256
|
+
judgedItems: number;
|
|
7257
|
+
topK: {
|
|
7258
|
+
yesWeight: number;
|
|
7259
|
+
totalWeight: number;
|
|
7260
|
+
};
|
|
7261
|
+
};
|
|
7262
|
+
/**
|
|
7263
|
+
* One per repetition that ran.
|
|
7264
|
+
*/
|
|
7265
|
+
candidate: Array<{
|
|
7266
|
+
yesWeight: number;
|
|
7267
|
+
totalWeight: number;
|
|
7268
|
+
items: number;
|
|
7269
|
+
judgedItems: number;
|
|
7270
|
+
topK: {
|
|
7271
|
+
yesWeight: number;
|
|
7272
|
+
totalWeight: number;
|
|
7273
|
+
};
|
|
7274
|
+
}>;
|
|
7275
|
+
/**
|
|
7276
|
+
* The first repetition's items against the judged ones.
|
|
7277
|
+
*/
|
|
7278
|
+
changes?: {
|
|
7279
|
+
kept: Array<{
|
|
7280
|
+
key: string;
|
|
7281
|
+
rankBefore?: number;
|
|
7282
|
+
rank?: number;
|
|
7283
|
+
}>;
|
|
7284
|
+
dropped: Array<{
|
|
7285
|
+
key: string;
|
|
7286
|
+
rankBefore?: number;
|
|
7287
|
+
}>;
|
|
7288
|
+
new: Array<{
|
|
7289
|
+
key: string;
|
|
7290
|
+
pointer: string;
|
|
7291
|
+
rank?: number;
|
|
7292
|
+
}>;
|
|
7293
|
+
};
|
|
7294
|
+
/**
|
|
7295
|
+
* The first repetition's tool calls, and what happened to each.
|
|
7296
|
+
*/
|
|
7297
|
+
tools?: Array<{
|
|
7298
|
+
step: number;
|
|
7299
|
+
callId: string;
|
|
7300
|
+
toolId: string;
|
|
7301
|
+
toolVersion: string;
|
|
7302
|
+
arguments: unknown;
|
|
7303
|
+
source: "live" | "recorded" | "refused";
|
|
7304
|
+
reason?: string;
|
|
7305
|
+
}>;
|
|
7306
|
+
diverged: boolean;
|
|
7307
|
+
refusedWrites: number;
|
|
7308
|
+
noContext: boolean;
|
|
7309
|
+
approvalSkipped: boolean;
|
|
7310
|
+
error?: string;
|
|
7311
|
+
/**
|
|
7312
|
+
* Set when the replay stopped at a refused write: what it would have done.
|
|
7313
|
+
*/
|
|
7314
|
+
stopped?: {
|
|
7315
|
+
toolId: string;
|
|
7316
|
+
arguments: unknown;
|
|
7317
|
+
reason?: string;
|
|
7318
|
+
};
|
|
7319
|
+
};
|
|
7320
|
+
/**
|
|
7321
|
+
* A comparison's `result` (a `judged` eval run's): the summary and each case. `EvalRun.result` stays an open object, since each kind has its own; the clients read it as this (TS `comparisonOf(run)`, Python `comparison_of(run)`).
|
|
7322
|
+
*/
|
|
7323
|
+
export type JudgedComparisonResult = {
|
|
7324
|
+
summary: JudgedComparisonSummary;
|
|
7325
|
+
perCase: Array<ComparisonCaseResult>;
|
|
7118
7326
|
};
|
|
7119
7327
|
export type EvalRun = {
|
|
7120
7328
|
runId: string;
|
|
@@ -7129,7 +7337,7 @@ declare namespace Schemas {
|
|
|
7129
7337
|
startedAt: string;
|
|
7130
7338
|
completedAt?: string;
|
|
7131
7339
|
/**
|
|
7132
|
-
* Kind-specific opaque JSON. For `accuracy`, contains `{ passCount, totalCount, meanScore, perCase[] }`. For `judged` (a comparison), `{ summary, perCase[] }`: the summary has the baseline (the versions behind the recorded runs) and the candidate (`{ kind: "agent", agentId, version }` or `{ kind: "flow", flowId, version }`), the case counts (`cases`, `diverged`, `refusedWrites`, `errors`, and `stopped`: flow cases that stopped at a write the replay refused, left out of the metrics), the models that answered, and `metrics` (`weightedYesShare`, `judgedCoverage`, `weightedPrecisionAtK`, each `{ baseline, candidate, delta, n, weight, baselineN, baselineWeight, direction, k?, spread? }`); each case has its replay runs, the scores, the items kept, dropped and new, the tool calls with what happened to each, and `stopped` (what it would have done) when it stopped. Other kinds define their own shapes as their dispatchers ship.
|
|
7340
|
+
* Kind-specific opaque JSON. For `accuracy`, contains `{ passCount, totalCount, meanScore, perCase[] }`. For `judged` (a comparison), `{ summary, perCase[] }`: the summary has the baseline (the versions behind the recorded runs) and the candidate (`{ kind: "agent", agentId, version }` or `{ kind: "flow", flowId, version, versions? }`), the case counts (`cases`, `diverged`, `refusedWrites`, `errors`, and `stopped`: flow cases that stopped at a write the replay refused, left out of the metrics), the models that answered, and `metrics` (`weightedYesShare`, `judgedCoverage`, `weightedPrecisionAtK`, each `{ baseline, candidate, delta, n, weight, baselineN, baselineWeight, direction, k?, spread? }`); each case has its replay runs, the scores, the items kept, dropped and new, the tool calls with what happened to each, and `stopped` (what it would have done) when it stopped. Other kinds define their own shapes as their dispatchers ship.
|
|
7133
7341
|
*/
|
|
7134
7342
|
result?: Record<string, unknown>;
|
|
7135
7343
|
error?: string;
|
|
@@ -7142,7 +7350,7 @@ declare namespace Schemas {
|
|
|
7142
7350
|
hasMore: boolean;
|
|
7143
7351
|
};
|
|
7144
7352
|
/**
|
|
7145
|
-
* Exactly one of `agentRef` or `flowRef` MUST be supplied. `dryRun: true` returns a plan preview without invoking the subject. For a `judged` suite (a test set), the run is a comparison: `agentRef` or `flowRef` with its `version` is the candidate, replayed on each case without doing anything the past run didn't (a flow stops at a write the replay refuses); `baseline` (default `'recorded'`), `reads` (default `recorded`), `repetitions` (default 1) and `k` (default 10) set how.
|
|
7353
|
+
* Exactly one of `agentRef` or `flowRef` MUST be supplied. `dryRun: true` returns a plan preview without invoking the subject. For a `judged` suite (a test set), the run is a comparison: `agentRef` or `flowRef` with its `version` is the candidate, replayed on each case without doing anything the past run didn't (a flow stops at a write the replay refuses); `baseline` (default `'recorded'`), `reads` (default `recorded`), `repetitions` (default 1) and `k` (default 10) set how. With `flowRef`, `versions` runs the flow with some of its agents or tools at other versions; an id the flow doesn't use, or a version that isn't published, is refused (`400 validation-failed`, each under `details.issues`).
|
|
7146
7354
|
*/
|
|
7147
7355
|
export type StartEvalRunBody = {
|
|
7148
7356
|
/**
|
|
@@ -7157,6 +7365,7 @@ declare namespace Schemas {
|
|
|
7157
7365
|
reads?: "recorded" | "live";
|
|
7158
7366
|
repetitions?: number;
|
|
7159
7367
|
k?: number;
|
|
7368
|
+
versions?: FlowVersionOverrides;
|
|
7160
7369
|
};
|
|
7161
7370
|
export type StartEvalRunResult = {
|
|
7162
7371
|
runId: string;
|
|
@@ -9019,6 +9228,383 @@ declare const EvalRunStatus: z.ZodEnum<{
|
|
|
9019
9228
|
completed: "completed";
|
|
9020
9229
|
cancelled: "cancelled";
|
|
9021
9230
|
}>;
|
|
9231
|
+
export type ComparisonMetric = Schemas.ComparisonMetric;
|
|
9232
|
+
export declare const ComparisonMetric: z.ZodObject<{
|
|
9233
|
+
baseline: z.ZodNullable<z.ZodNumber>;
|
|
9234
|
+
candidate: z.ZodNullable<z.ZodNumber>;
|
|
9235
|
+
delta: z.ZodNullable<z.ZodNumber>;
|
|
9236
|
+
n: z.ZodNumber;
|
|
9237
|
+
weight: z.ZodNumber;
|
|
9238
|
+
baselineN: z.ZodNumber;
|
|
9239
|
+
baselineWeight: z.ZodNumber;
|
|
9240
|
+
direction: z.ZodLiteral<"higher">;
|
|
9241
|
+
k: z.ZodOptional<z.ZodNumber>;
|
|
9242
|
+
spread: z.ZodOptional<z.ZodNumber>;
|
|
9243
|
+
}, z.core.$strict>;
|
|
9244
|
+
export type ComparisonCandidate = Schemas.ComparisonCandidate;
|
|
9245
|
+
export declare const ComparisonCandidate: z.ZodUnion<readonly [
|
|
9246
|
+
z.ZodObject<{
|
|
9247
|
+
kind: z.ZodLiteral<"agent">;
|
|
9248
|
+
agentId: z.ZodString;
|
|
9249
|
+
version: z.ZodString;
|
|
9250
|
+
}, z.core.$strict>,
|
|
9251
|
+
z.ZodObject<{
|
|
9252
|
+
kind: z.ZodLiteral<"flow">;
|
|
9253
|
+
flowId: z.ZodString;
|
|
9254
|
+
version: z.ZodString;
|
|
9255
|
+
versions: z.ZodOptional<z.ZodObject<{
|
|
9256
|
+
tools: z.ZodOptional<z.ZodRecord<z.ZodString, z.ZodString>>;
|
|
9257
|
+
agents: z.ZodOptional<z.ZodRecord<z.ZodString, z.ZodString>>;
|
|
9258
|
+
}, z.core.$strict>>;
|
|
9259
|
+
}, z.core.$strict>
|
|
9260
|
+
]>;
|
|
9261
|
+
export type JudgedComparisonSummary = Schemas.JudgedComparisonSummary;
|
|
9262
|
+
export declare const JudgedComparisonSummary: z.ZodObject<{
|
|
9263
|
+
evalRunId: z.ZodString;
|
|
9264
|
+
status: z.ZodEnum<{
|
|
9265
|
+
failed: "failed";
|
|
9266
|
+
completed: "completed";
|
|
9267
|
+
partial: "partial";
|
|
9268
|
+
}>;
|
|
9269
|
+
completedAt: z.ZodISODateTime;
|
|
9270
|
+
suite: z.ZodObject<{
|
|
9271
|
+
id: z.ZodString;
|
|
9272
|
+
version: z.ZodString;
|
|
9273
|
+
}, z.core.$strict>;
|
|
9274
|
+
candidate: z.ZodUnion<readonly [
|
|
9275
|
+
z.ZodObject<{
|
|
9276
|
+
kind: z.ZodLiteral<"agent">;
|
|
9277
|
+
agentId: z.ZodString;
|
|
9278
|
+
version: z.ZodString;
|
|
9279
|
+
}, z.core.$strict>,
|
|
9280
|
+
z.ZodObject<{
|
|
9281
|
+
kind: z.ZodLiteral<"flow">;
|
|
9282
|
+
flowId: z.ZodString;
|
|
9283
|
+
version: z.ZodString;
|
|
9284
|
+
versions: z.ZodOptional<z.ZodObject<{
|
|
9285
|
+
tools: z.ZodOptional<z.ZodRecord<z.ZodString, z.ZodString>>;
|
|
9286
|
+
agents: z.ZodOptional<z.ZodRecord<z.ZodString, z.ZodString>>;
|
|
9287
|
+
}, z.core.$strict>>;
|
|
9288
|
+
}, z.core.$strict>
|
|
9289
|
+
]>;
|
|
9290
|
+
baseline: z.ZodUnion<readonly [
|
|
9291
|
+
z.ZodObject<{
|
|
9292
|
+
kind: z.ZodLiteral<"recorded">;
|
|
9293
|
+
versions: z.ZodArray<z.ZodObject<{
|
|
9294
|
+
agentId: z.ZodOptional<z.ZodString>;
|
|
9295
|
+
flowId: z.ZodOptional<z.ZodString>;
|
|
9296
|
+
version: z.ZodString;
|
|
9297
|
+
cases: z.ZodNumber;
|
|
9298
|
+
}, z.core.$strict>>;
|
|
9299
|
+
}, z.core.$strict>,
|
|
9300
|
+
z.ZodObject<{
|
|
9301
|
+
kind: z.ZodLiteral<"version">;
|
|
9302
|
+
agentId: z.ZodString;
|
|
9303
|
+
version: z.ZodString;
|
|
9304
|
+
via: z.ZodEnum<{
|
|
9305
|
+
live: "live";
|
|
9306
|
+
explicit: "explicit";
|
|
9307
|
+
}>;
|
|
9308
|
+
liveScope: z.ZodOptional<z.ZodRecord<z.ZodString, z.ZodUnknown>>;
|
|
9309
|
+
}, z.core.$strict>
|
|
9310
|
+
]>;
|
|
9311
|
+
scope: z.ZodObject<{
|
|
9312
|
+
projectId: z.ZodOptional<z.ZodString>;
|
|
9313
|
+
}, z.core.$strict>;
|
|
9314
|
+
cases: z.ZodNumber;
|
|
9315
|
+
diverged: z.ZodNumber;
|
|
9316
|
+
refusedWrites: z.ZodNumber;
|
|
9317
|
+
errors: z.ZodNumber;
|
|
9318
|
+
stopped: z.ZodNumber;
|
|
9319
|
+
reads: z.ZodEnum<{
|
|
9320
|
+
recorded: "recorded";
|
|
9321
|
+
live: "live";
|
|
9322
|
+
}>;
|
|
9323
|
+
sampling: z.ZodObject<{
|
|
9324
|
+
models: z.ZodArray<z.ZodObject<{
|
|
9325
|
+
providerId: z.ZodString;
|
|
9326
|
+
model: z.ZodString;
|
|
9327
|
+
runs: z.ZodNumber;
|
|
9328
|
+
}, z.core.$strict>>;
|
|
9329
|
+
}, z.core.$strict>;
|
|
9330
|
+
repetitions: z.ZodNumber;
|
|
9331
|
+
metrics: z.ZodObject<{
|
|
9332
|
+
weightedYesShare: z.ZodObject<{
|
|
9333
|
+
baseline: z.ZodNullable<z.ZodNumber>;
|
|
9334
|
+
candidate: z.ZodNullable<z.ZodNumber>;
|
|
9335
|
+
delta: z.ZodNullable<z.ZodNumber>;
|
|
9336
|
+
n: z.ZodNumber;
|
|
9337
|
+
weight: z.ZodNumber;
|
|
9338
|
+
baselineN: z.ZodNumber;
|
|
9339
|
+
baselineWeight: z.ZodNumber;
|
|
9340
|
+
direction: z.ZodLiteral<"higher">;
|
|
9341
|
+
k: z.ZodOptional<z.ZodNumber>;
|
|
9342
|
+
spread: z.ZodOptional<z.ZodNumber>;
|
|
9343
|
+
}, z.core.$strict>;
|
|
9344
|
+
judgedCoverage: z.ZodObject<{
|
|
9345
|
+
baseline: z.ZodNullable<z.ZodNumber>;
|
|
9346
|
+
candidate: z.ZodNullable<z.ZodNumber>;
|
|
9347
|
+
delta: z.ZodNullable<z.ZodNumber>;
|
|
9348
|
+
n: z.ZodNumber;
|
|
9349
|
+
weight: z.ZodNumber;
|
|
9350
|
+
baselineN: z.ZodNumber;
|
|
9351
|
+
baselineWeight: z.ZodNumber;
|
|
9352
|
+
direction: z.ZodLiteral<"higher">;
|
|
9353
|
+
k: z.ZodOptional<z.ZodNumber>;
|
|
9354
|
+
spread: z.ZodOptional<z.ZodNumber>;
|
|
9355
|
+
}, z.core.$strict>;
|
|
9356
|
+
weightedPrecisionAtK: z.ZodObject<{
|
|
9357
|
+
baseline: z.ZodNullable<z.ZodNumber>;
|
|
9358
|
+
candidate: z.ZodNullable<z.ZodNumber>;
|
|
9359
|
+
delta: z.ZodNullable<z.ZodNumber>;
|
|
9360
|
+
n: z.ZodNumber;
|
|
9361
|
+
weight: z.ZodNumber;
|
|
9362
|
+
baselineN: z.ZodNumber;
|
|
9363
|
+
baselineWeight: z.ZodNumber;
|
|
9364
|
+
direction: z.ZodLiteral<"higher">;
|
|
9365
|
+
k: z.ZodOptional<z.ZodNumber>;
|
|
9366
|
+
spread: z.ZodOptional<z.ZodNumber>;
|
|
9367
|
+
}, z.core.$strict>;
|
|
9368
|
+
}, z.core.$strict>;
|
|
9369
|
+
}, z.core.$strict>;
|
|
9370
|
+
export type ComparisonCaseResult = Schemas.ComparisonCaseResult;
|
|
9371
|
+
export declare const ComparisonCaseResult: z.ZodObject<{
|
|
9372
|
+
caseId: z.ZodString;
|
|
9373
|
+
runIds: z.ZodArray<z.ZodString>;
|
|
9374
|
+
baseline: z.ZodObject<{
|
|
9375
|
+
yesWeight: z.ZodNumber;
|
|
9376
|
+
totalWeight: z.ZodNumber;
|
|
9377
|
+
items: z.ZodNumber;
|
|
9378
|
+
judgedItems: z.ZodNumber;
|
|
9379
|
+
topK: z.ZodObject<{
|
|
9380
|
+
yesWeight: z.ZodNumber;
|
|
9381
|
+
totalWeight: z.ZodNumber;
|
|
9382
|
+
}, z.core.$strict>;
|
|
9383
|
+
}, z.core.$strict>;
|
|
9384
|
+
candidate: z.ZodArray<z.ZodObject<{
|
|
9385
|
+
yesWeight: z.ZodNumber;
|
|
9386
|
+
totalWeight: z.ZodNumber;
|
|
9387
|
+
items: z.ZodNumber;
|
|
9388
|
+
judgedItems: z.ZodNumber;
|
|
9389
|
+
topK: z.ZodObject<{
|
|
9390
|
+
yesWeight: z.ZodNumber;
|
|
9391
|
+
totalWeight: z.ZodNumber;
|
|
9392
|
+
}, z.core.$strict>;
|
|
9393
|
+
}, z.core.$strict>>;
|
|
9394
|
+
changes: z.ZodOptional<z.ZodObject<{
|
|
9395
|
+
kept: z.ZodArray<z.ZodObject<{
|
|
9396
|
+
key: z.ZodString;
|
|
9397
|
+
rankBefore: z.ZodOptional<z.ZodNumber>;
|
|
9398
|
+
rank: z.ZodOptional<z.ZodNumber>;
|
|
9399
|
+
}, z.core.$strict>>;
|
|
9400
|
+
dropped: z.ZodArray<z.ZodObject<{
|
|
9401
|
+
key: z.ZodString;
|
|
9402
|
+
rankBefore: z.ZodOptional<z.ZodNumber>;
|
|
9403
|
+
}, z.core.$strict>>;
|
|
9404
|
+
new: z.ZodArray<z.ZodObject<{
|
|
9405
|
+
key: z.ZodString;
|
|
9406
|
+
pointer: z.ZodString;
|
|
9407
|
+
rank: z.ZodOptional<z.ZodNumber>;
|
|
9408
|
+
}, z.core.$strict>>;
|
|
9409
|
+
}, z.core.$strict>>;
|
|
9410
|
+
tools: z.ZodOptional<z.ZodArray<z.ZodObject<{
|
|
9411
|
+
step: z.ZodNumber;
|
|
9412
|
+
callId: z.ZodString;
|
|
9413
|
+
toolId: z.ZodString;
|
|
9414
|
+
toolVersion: z.ZodString;
|
|
9415
|
+
arguments: z.ZodUnknown;
|
|
9416
|
+
source: z.ZodEnum<{
|
|
9417
|
+
recorded: "recorded";
|
|
9418
|
+
live: "live";
|
|
9419
|
+
refused: "refused";
|
|
9420
|
+
}>;
|
|
9421
|
+
reason: z.ZodOptional<z.ZodString>;
|
|
9422
|
+
}, z.core.$strict>>>;
|
|
9423
|
+
diverged: z.ZodBoolean;
|
|
9424
|
+
refusedWrites: z.ZodNumber;
|
|
9425
|
+
noContext: z.ZodBoolean;
|
|
9426
|
+
approvalSkipped: z.ZodBoolean;
|
|
9427
|
+
error: z.ZodOptional<z.ZodString>;
|
|
9428
|
+
stopped: z.ZodOptional<z.ZodObject<{
|
|
9429
|
+
toolId: z.ZodString;
|
|
9430
|
+
arguments: z.ZodUnknown;
|
|
9431
|
+
reason: z.ZodOptional<z.ZodString>;
|
|
9432
|
+
}, z.core.$strict>>;
|
|
9433
|
+
}, z.core.$strict>;
|
|
9434
|
+
export type JudgedComparisonResult = Schemas.JudgedComparisonResult;
|
|
9435
|
+
export declare const JudgedComparisonResult: z.ZodObject<{
|
|
9436
|
+
summary: z.ZodObject<{
|
|
9437
|
+
evalRunId: z.ZodString;
|
|
9438
|
+
status: z.ZodEnum<{
|
|
9439
|
+
failed: "failed";
|
|
9440
|
+
completed: "completed";
|
|
9441
|
+
partial: "partial";
|
|
9442
|
+
}>;
|
|
9443
|
+
completedAt: z.ZodISODateTime;
|
|
9444
|
+
suite: z.ZodObject<{
|
|
9445
|
+
id: z.ZodString;
|
|
9446
|
+
version: z.ZodString;
|
|
9447
|
+
}, z.core.$strict>;
|
|
9448
|
+
candidate: z.ZodUnion<readonly [
|
|
9449
|
+
z.ZodObject<{
|
|
9450
|
+
kind: z.ZodLiteral<"agent">;
|
|
9451
|
+
agentId: z.ZodString;
|
|
9452
|
+
version: z.ZodString;
|
|
9453
|
+
}, z.core.$strict>,
|
|
9454
|
+
z.ZodObject<{
|
|
9455
|
+
kind: z.ZodLiteral<"flow">;
|
|
9456
|
+
flowId: z.ZodString;
|
|
9457
|
+
version: z.ZodString;
|
|
9458
|
+
versions: z.ZodOptional<z.ZodObject<{
|
|
9459
|
+
tools: z.ZodOptional<z.ZodRecord<z.ZodString, z.ZodString>>;
|
|
9460
|
+
agents: z.ZodOptional<z.ZodRecord<z.ZodString, z.ZodString>>;
|
|
9461
|
+
}, z.core.$strict>>;
|
|
9462
|
+
}, z.core.$strict>
|
|
9463
|
+
]>;
|
|
9464
|
+
baseline: z.ZodUnion<readonly [
|
|
9465
|
+
z.ZodObject<{
|
|
9466
|
+
kind: z.ZodLiteral<"recorded">;
|
|
9467
|
+
versions: z.ZodArray<z.ZodObject<{
|
|
9468
|
+
agentId: z.ZodOptional<z.ZodString>;
|
|
9469
|
+
flowId: z.ZodOptional<z.ZodString>;
|
|
9470
|
+
version: z.ZodString;
|
|
9471
|
+
cases: z.ZodNumber;
|
|
9472
|
+
}, z.core.$strict>>;
|
|
9473
|
+
}, z.core.$strict>,
|
|
9474
|
+
z.ZodObject<{
|
|
9475
|
+
kind: z.ZodLiteral<"version">;
|
|
9476
|
+
agentId: z.ZodString;
|
|
9477
|
+
version: z.ZodString;
|
|
9478
|
+
via: z.ZodEnum<{
|
|
9479
|
+
live: "live";
|
|
9480
|
+
explicit: "explicit";
|
|
9481
|
+
}>;
|
|
9482
|
+
liveScope: z.ZodOptional<z.ZodRecord<z.ZodString, z.ZodUnknown>>;
|
|
9483
|
+
}, z.core.$strict>
|
|
9484
|
+
]>;
|
|
9485
|
+
scope: z.ZodObject<{
|
|
9486
|
+
projectId: z.ZodOptional<z.ZodString>;
|
|
9487
|
+
}, z.core.$strict>;
|
|
9488
|
+
cases: z.ZodNumber;
|
|
9489
|
+
diverged: z.ZodNumber;
|
|
9490
|
+
refusedWrites: z.ZodNumber;
|
|
9491
|
+
errors: z.ZodNumber;
|
|
9492
|
+
stopped: z.ZodNumber;
|
|
9493
|
+
reads: z.ZodEnum<{
|
|
9494
|
+
recorded: "recorded";
|
|
9495
|
+
live: "live";
|
|
9496
|
+
}>;
|
|
9497
|
+
sampling: z.ZodObject<{
|
|
9498
|
+
models: z.ZodArray<z.ZodObject<{
|
|
9499
|
+
providerId: z.ZodString;
|
|
9500
|
+
model: z.ZodString;
|
|
9501
|
+
runs: z.ZodNumber;
|
|
9502
|
+
}, z.core.$strict>>;
|
|
9503
|
+
}, z.core.$strict>;
|
|
9504
|
+
repetitions: z.ZodNumber;
|
|
9505
|
+
metrics: z.ZodObject<{
|
|
9506
|
+
weightedYesShare: z.ZodObject<{
|
|
9507
|
+
baseline: z.ZodNullable<z.ZodNumber>;
|
|
9508
|
+
candidate: z.ZodNullable<z.ZodNumber>;
|
|
9509
|
+
delta: z.ZodNullable<z.ZodNumber>;
|
|
9510
|
+
n: z.ZodNumber;
|
|
9511
|
+
weight: z.ZodNumber;
|
|
9512
|
+
baselineN: z.ZodNumber;
|
|
9513
|
+
baselineWeight: z.ZodNumber;
|
|
9514
|
+
direction: z.ZodLiteral<"higher">;
|
|
9515
|
+
k: z.ZodOptional<z.ZodNumber>;
|
|
9516
|
+
spread: z.ZodOptional<z.ZodNumber>;
|
|
9517
|
+
}, z.core.$strict>;
|
|
9518
|
+
judgedCoverage: z.ZodObject<{
|
|
9519
|
+
baseline: z.ZodNullable<z.ZodNumber>;
|
|
9520
|
+
candidate: z.ZodNullable<z.ZodNumber>;
|
|
9521
|
+
delta: z.ZodNullable<z.ZodNumber>;
|
|
9522
|
+
n: z.ZodNumber;
|
|
9523
|
+
weight: z.ZodNumber;
|
|
9524
|
+
baselineN: z.ZodNumber;
|
|
9525
|
+
baselineWeight: z.ZodNumber;
|
|
9526
|
+
direction: z.ZodLiteral<"higher">;
|
|
9527
|
+
k: z.ZodOptional<z.ZodNumber>;
|
|
9528
|
+
spread: z.ZodOptional<z.ZodNumber>;
|
|
9529
|
+
}, z.core.$strict>;
|
|
9530
|
+
weightedPrecisionAtK: z.ZodObject<{
|
|
9531
|
+
baseline: z.ZodNullable<z.ZodNumber>;
|
|
9532
|
+
candidate: z.ZodNullable<z.ZodNumber>;
|
|
9533
|
+
delta: z.ZodNullable<z.ZodNumber>;
|
|
9534
|
+
n: z.ZodNumber;
|
|
9535
|
+
weight: z.ZodNumber;
|
|
9536
|
+
baselineN: z.ZodNumber;
|
|
9537
|
+
baselineWeight: z.ZodNumber;
|
|
9538
|
+
direction: z.ZodLiteral<"higher">;
|
|
9539
|
+
k: z.ZodOptional<z.ZodNumber>;
|
|
9540
|
+
spread: z.ZodOptional<z.ZodNumber>;
|
|
9541
|
+
}, z.core.$strict>;
|
|
9542
|
+
}, z.core.$strict>;
|
|
9543
|
+
}, z.core.$strict>;
|
|
9544
|
+
perCase: z.ZodArray<z.ZodObject<{
|
|
9545
|
+
caseId: z.ZodString;
|
|
9546
|
+
runIds: z.ZodArray<z.ZodString>;
|
|
9547
|
+
baseline: z.ZodObject<{
|
|
9548
|
+
yesWeight: z.ZodNumber;
|
|
9549
|
+
totalWeight: z.ZodNumber;
|
|
9550
|
+
items: z.ZodNumber;
|
|
9551
|
+
judgedItems: z.ZodNumber;
|
|
9552
|
+
topK: z.ZodObject<{
|
|
9553
|
+
yesWeight: z.ZodNumber;
|
|
9554
|
+
totalWeight: z.ZodNumber;
|
|
9555
|
+
}, z.core.$strict>;
|
|
9556
|
+
}, z.core.$strict>;
|
|
9557
|
+
candidate: z.ZodArray<z.ZodObject<{
|
|
9558
|
+
yesWeight: z.ZodNumber;
|
|
9559
|
+
totalWeight: z.ZodNumber;
|
|
9560
|
+
items: z.ZodNumber;
|
|
9561
|
+
judgedItems: z.ZodNumber;
|
|
9562
|
+
topK: z.ZodObject<{
|
|
9563
|
+
yesWeight: z.ZodNumber;
|
|
9564
|
+
totalWeight: z.ZodNumber;
|
|
9565
|
+
}, z.core.$strict>;
|
|
9566
|
+
}, z.core.$strict>>;
|
|
9567
|
+
changes: z.ZodOptional<z.ZodObject<{
|
|
9568
|
+
kept: z.ZodArray<z.ZodObject<{
|
|
9569
|
+
key: z.ZodString;
|
|
9570
|
+
rankBefore: z.ZodOptional<z.ZodNumber>;
|
|
9571
|
+
rank: z.ZodOptional<z.ZodNumber>;
|
|
9572
|
+
}, z.core.$strict>>;
|
|
9573
|
+
dropped: z.ZodArray<z.ZodObject<{
|
|
9574
|
+
key: z.ZodString;
|
|
9575
|
+
rankBefore: z.ZodOptional<z.ZodNumber>;
|
|
9576
|
+
}, z.core.$strict>>;
|
|
9577
|
+
new: z.ZodArray<z.ZodObject<{
|
|
9578
|
+
key: z.ZodString;
|
|
9579
|
+
pointer: z.ZodString;
|
|
9580
|
+
rank: z.ZodOptional<z.ZodNumber>;
|
|
9581
|
+
}, z.core.$strict>>;
|
|
9582
|
+
}, z.core.$strict>>;
|
|
9583
|
+
tools: z.ZodOptional<z.ZodArray<z.ZodObject<{
|
|
9584
|
+
step: z.ZodNumber;
|
|
9585
|
+
callId: z.ZodString;
|
|
9586
|
+
toolId: z.ZodString;
|
|
9587
|
+
toolVersion: z.ZodString;
|
|
9588
|
+
arguments: z.ZodUnknown;
|
|
9589
|
+
source: z.ZodEnum<{
|
|
9590
|
+
recorded: "recorded";
|
|
9591
|
+
live: "live";
|
|
9592
|
+
refused: "refused";
|
|
9593
|
+
}>;
|
|
9594
|
+
reason: z.ZodOptional<z.ZodString>;
|
|
9595
|
+
}, z.core.$strict>>>;
|
|
9596
|
+
diverged: z.ZodBoolean;
|
|
9597
|
+
refusedWrites: z.ZodNumber;
|
|
9598
|
+
noContext: z.ZodBoolean;
|
|
9599
|
+
approvalSkipped: z.ZodBoolean;
|
|
9600
|
+
error: z.ZodOptional<z.ZodString>;
|
|
9601
|
+
stopped: z.ZodOptional<z.ZodObject<{
|
|
9602
|
+
toolId: z.ZodString;
|
|
9603
|
+
arguments: z.ZodUnknown;
|
|
9604
|
+
reason: z.ZodOptional<z.ZodString>;
|
|
9605
|
+
}, z.core.$strict>>;
|
|
9606
|
+
}, z.core.$strict>>;
|
|
9607
|
+
}, z.core.$strict>;
|
|
9022
9608
|
export type EvalRun = Schemas.EvalRun;
|
|
9023
9609
|
declare const EvalRun: z.ZodObject<{
|
|
9024
9610
|
runId: z.ZodUUID;
|
|
@@ -9075,6 +9661,10 @@ declare const EvalRun: z.ZodObject<{
|
|
|
9075
9661
|
}>;
|
|
9076
9662
|
repetitions: z.ZodNumber;
|
|
9077
9663
|
k: z.ZodNumber;
|
|
9664
|
+
versions: z.ZodOptional<z.ZodObject<{
|
|
9665
|
+
tools: z.ZodOptional<z.ZodRecord<z.ZodString, z.ZodString>>;
|
|
9666
|
+
agents: z.ZodOptional<z.ZodRecord<z.ZodString, z.ZodString>>;
|
|
9667
|
+
}, z.core.$strict>>;
|
|
9078
9668
|
}, z.core.$strict>>;
|
|
9079
9669
|
}, z.core.$strict>;
|
|
9080
9670
|
export type EvalRunCollectionPage = Schemas.EvalRunCollectionPage;
|
|
@@ -9134,6 +9724,10 @@ declare const EvalRunCollectionPage: z.ZodObject<{
|
|
|
9134
9724
|
}>;
|
|
9135
9725
|
repetitions: z.ZodNumber;
|
|
9136
9726
|
k: z.ZodNumber;
|
|
9727
|
+
versions: z.ZodOptional<z.ZodObject<{
|
|
9728
|
+
tools: z.ZodOptional<z.ZodRecord<z.ZodString, z.ZodString>>;
|
|
9729
|
+
agents: z.ZodOptional<z.ZodRecord<z.ZodString, z.ZodString>>;
|
|
9730
|
+
}, z.core.$strict>>;
|
|
9137
9731
|
}, z.core.$strict>>;
|
|
9138
9732
|
}, z.core.$strict>>;
|
|
9139
9733
|
nextCursor: z.ZodOptional<z.ZodString>;
|
|
@@ -9171,6 +9765,10 @@ declare const StartEvalRunBody: z.ZodObject<{
|
|
|
9171
9765
|
}>>;
|
|
9172
9766
|
repetitions: z.ZodOptional<z.ZodNumber>;
|
|
9173
9767
|
k: z.ZodOptional<z.ZodNumber>;
|
|
9768
|
+
versions: z.ZodOptional<z.ZodObject<{
|
|
9769
|
+
tools: z.ZodOptional<z.ZodRecord<z.ZodString, z.ZodString>>;
|
|
9770
|
+
agents: z.ZodOptional<z.ZodRecord<z.ZodString, z.ZodString>>;
|
|
9771
|
+
}, z.core.$strict>>;
|
|
9174
9772
|
}, z.core.$strict>;
|
|
9175
9773
|
export type StartEvalRunResult = Schemas.StartEvalRunResult;
|
|
9176
9774
|
declare const StartEvalRunResult: z.ZodObject<{
|
|
@@ -11338,6 +11936,13 @@ export type EvalRunPage = EvalRunCollectionPage;
|
|
|
11338
11936
|
*/
|
|
11339
11937
|
export type StartEvalRunInput = Omit<StartEvalRunBody, "projectId">;
|
|
11340
11938
|
export type StartEvalRunOutcome = StartEvalRunResult;
|
|
11939
|
+
/**
|
|
11940
|
+
* A comparison's result, typed from the OpenAPI `JudgedComparisonResult`:
|
|
11941
|
+
* its summary and each case. `run` is a `judged` eval run (a test set
|
|
11942
|
+
* compared with a version); `undefined` for another kind of eval run, a
|
|
11943
|
+
* dry run, or one that hasn't finished.
|
|
11944
|
+
*/
|
|
11945
|
+
export declare function comparisonOf(run: EvalRunRecord): JudgedComparisonResult | undefined;
|
|
11341
11946
|
export interface StartEvalRunOptions {
|
|
11342
11947
|
/** Project the eval run belongs to. The route requires it. */
|
|
11343
11948
|
readonly projectId: string;
|