@kindgi/client 0.1.3 → 0.1.4-rc.1

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/dist/index.d.cts CHANGED
@@ -218,8 +218,54 @@ export interface Page<T> {
218
218
  * The brand prevents mixing free-form strings with validated semver values.
219
219
  */
220
220
  export type Semver = Brand<string, "Semver">;
221
+ /**
222
+ * Why the runtime registered a version of an agent or flow under another
223
+ * number than its definition names. Versions never change, so a deploy
224
+ * whose definition's version is registered already with other content
225
+ * registers the next free version in its line instead:
226
+ *
227
+ * - `pins-changed`: the definition's version is registered with other
228
+ * pins (a block it uses has a new version);
229
+ * - `unpinned`: the definition's version was published before pins
230
+ * existed, so it has none;
231
+ * - `version-taken`: the definition's version holds another definition;
232
+ * - `edited`: derived from an existing version with some data-block pins
233
+ * swapped (an expert's edit), not from a definition at all.
234
+ */
235
+ export type VersionDerivationReason = "pins-changed" | "unpinned" | "version-taken" | "edited";
236
+ /** The version a version was registered in place of (or derived from), and why. */
237
+ export interface VersionDerivation {
238
+ /** The version the definition names, or the one an edit derived from. */
239
+ readonly version: string;
240
+ readonly reason: VersionDerivationReason;
241
+ /** For an edit: a short label for the version (e.g. "warmer tone"). */
242
+ readonly label?: string;
243
+ /** For an edit: who made it (`user:<id>`, or the token for a key). */
244
+ readonly by?: string;
245
+ }
221
246
  /** Terminal states are `completed`, `failed`, `cancelled`. */
222
247
  export type RunStatus = "pending" | "running" | "suspended" | "completed" | "failed" | "cancelled";
248
+ /**
249
+ * The exact version of each block a flow version runs: its lockfile.
250
+ *
251
+ * A flow names its tools by id (a tool node's `ref`, a fanout branch's
252
+ * `handler`) and its agents by id, with an exact `config.version` or
253
+ * none. When a version of the flow is published, the runtime pins each
254
+ * tool, and each agent that names no version, to its latest version
255
+ * then, and every run of that flow version uses those versions. So a
256
+ * new tool or agent version reaches the flow only through a new flow
257
+ * version.
258
+ *
259
+ * Set by the runtime at publish, never authored. A flow version
260
+ * published before pins existed has none and binds the latest versions
261
+ * per run.
262
+ */
263
+ export interface FlowPins {
264
+ /** Tool id → the exact version this flow version runs. */
265
+ readonly tools: Readonly<Record<string, string>>;
266
+ /** Agent id → the exact version its agent nodes that name none run. */
267
+ readonly agents: Readonly<Record<string, string>>;
268
+ }
223
269
  /**
224
270
  * A path into the run's addressable state. Dot-separated segments, always
225
271
  * starting with a recognized root. Path syntax is deliberately narrow — no
@@ -928,6 +974,20 @@ export interface Flow {
928
974
  * `$end` edge (the lowest-id taken one when several fire).
929
975
  */
930
976
  readonly output?: FlowOutputSpec;
977
+ /**
978
+ * The exact tool and agent versions this flow version runs, resolved by
979
+ * the runtime when the version was published (see `FlowPins`). Never
980
+ * authored: `loadFlow` refuses it. Absent on a flow version published
981
+ * before pins existed; it binds the latest versions per run.
982
+ */
983
+ readonly pins?: FlowPins;
984
+ /** `flowPinsDigest(pins)`, recorded when the version was published. */
985
+ readonly pinsDigest?: string;
986
+ /**
987
+ * Set by the runtime on a version a deploy registered under another
988
+ * number than the definition's (see `VersionDerivation`). Never authored.
989
+ */
990
+ readonly derivedFrom?: VersionDerivation;
931
991
  }
932
992
  /**
933
993
  * A flow's declared output: what the run returns, built from the
@@ -1172,6 +1232,29 @@ export interface ToolHitlRule {
1172
1232
  readonly mode: ToolHitlMode;
1173
1233
  readonly requiredRole?: ReviewerRole;
1174
1234
  }
1235
+ /**
1236
+ * The exact version of each block an agent version runs: its lockfile.
1237
+ *
1238
+ * An agent names its tools by range (`{ id: 'acme.lookup', version:
1239
+ * '^1.0.0' }`), like `package.json`. When a version of the agent is
1240
+ * published, the runtime resolves each range once, and every run of that
1241
+ * version uses the versions recorded here. So a new tool version reaches
1242
+ * an agent only through a new agent version, and two runs of one agent
1243
+ * version always run the same blocks.
1244
+ *
1245
+ * Set by the runtime at publish, never authored. A version published
1246
+ * before pins existed has none and resolves its ranges per run.
1247
+ */
1248
+ export interface AgentPins {
1249
+ /** Tool id → the exact version this agent version runs. */
1250
+ readonly tools: Readonly<Record<string, string>>;
1251
+ /** Prompt block id → exact version. Empty until the agent references prompt blocks. */
1252
+ readonly prompts: Readonly<Record<string, string>>;
1253
+ /** Settings block id → exact version. Empty until the agent references settings blocks. */
1254
+ readonly settings: Readonly<Record<string, string>>;
1255
+ }
1256
+ /** The version an agent version was registered in place of, and why. */
1257
+ export type AgentDerivation = VersionDerivation;
1175
1258
  type AgentId$1 = Brand<string, "AgentId">;
1176
1259
  /**
1177
1260
  * A single message inside a conversation. Persisted as a `Fact` in
@@ -1269,6 +1352,20 @@ export interface ToolRef {
1269
1352
  readonly id: string;
1270
1353
  readonly version: string;
1271
1354
  }
1355
+ /**
1356
+ * A prompt block an agent's instructions come from: its id and a semver
1357
+ * range, resolved like a tool's (`pickVersion`) and pinned when the
1358
+ * agent version is published.
1359
+ */
1360
+ export interface PromptRef {
1361
+ readonly prompt: string;
1362
+ readonly version: string;
1363
+ }
1364
+ /** A settings block an agent reads: its id and a semver range, pinned at publish. */
1365
+ export interface BlockRef {
1366
+ readonly id: string;
1367
+ readonly version: string;
1368
+ }
1272
1369
  /**
1273
1370
  * Per-agent behavior for multi-turn conversations. The agent chooses:
1274
1371
  * - How many prior messages to load (`historyLimit`; unset = all).
@@ -1413,8 +1510,13 @@ export interface Agent {
1413
1510
  * unresolved reference fails the turn at invoke time
1414
1511
  * (`model-invocation-failed` whose `cause` is the `missing-parameter`
1415
1512
  * render error), never a silent empty string.
1513
+ *
1514
+ * Or a prompt block, by range (`{ prompt: 'acme.intake-prompt',
1515
+ * version: '^1.0.0' }`): its template renders here instead, with the
1516
+ * parameters it declares, and the version that runs is pinned when the
1517
+ * agent version is published (`pins.prompts`).
1416
1518
  */
1417
- readonly instructions: string;
1519
+ readonly instructions: string | PromptRef;
1418
1520
  /**
1419
1521
  * Typed parameters the caller supplies at invoke time. The UI reads
1420
1522
  * this to build a "configure agent" form; the runtime validates each
@@ -1526,6 +1628,35 @@ export interface Agent {
1526
1628
  * A tenant's `tool-errors` policy can lower it. See `ToolErrorsSpec`.
1527
1629
  */
1528
1630
  readonly toolErrors?: ToolErrorsSpec;
1631
+ /**
1632
+ * The exact block versions this agent version runs, resolved by the
1633
+ * runtime when the version was published (see `AgentPins`). Never
1634
+ * authored: `defineAgent` doesn't take it. Absent on an agent defined
1635
+ * in code and on a version published before pins existed; its tool
1636
+ * ranges then resolve per run.
1637
+ */
1638
+ readonly pins?: AgentPins;
1639
+ /**
1640
+ * Settings blocks the agent reads, by range. Each block's values reach
1641
+ * its tools as `ToolContext.settings[<block id>]` and its templates as
1642
+ * `settings.<block id>.<key>`; the versions are pinned at publish
1643
+ * (`pins.settings`).
1644
+ */
1645
+ readonly settings?: readonly BlockRef[];
1646
+ /**
1647
+ * A settings block of model settings (`MODEL_SETTINGS_SCHEMA`:
1648
+ * `temperature`, `maxOutputTokens`) the turn's model calls use. Pinned
1649
+ * at publish with the other settings.
1650
+ */
1651
+ readonly modelSettings?: BlockRef;
1652
+ /** `pinsDigest(pins)`, recorded when the version was published. */
1653
+ readonly pinsDigest?: string;
1654
+ /**
1655
+ * Set by the runtime on a version it registered under another number
1656
+ * than the definition's, when a deploy couldn't register that number
1657
+ * as it was (see `AgentDerivation`). Never authored.
1658
+ */
1659
+ readonly derivedFrom?: AgentDerivation;
1529
1660
  }
1530
1661
  /** A retrieved fact + the retrieval intent that pulled it. */
1531
1662
  export interface RetrievedFact {
@@ -1616,8 +1747,16 @@ export interface DefineAgentSpec {
1616
1747
  * The system prompt. Sent to the model with every turn as the
1617
1748
  * baseline instructions. Load-bearing — this is where you shape
1618
1749
  * the agent's behavior (persona, output format, tool-use policy).
1750
+ *
1751
+ * Or a prompt block by range (`{ prompt: 'acme.intake-prompt',
1752
+ * version: '^1.0.0' }`), whose template and parameters are used
1753
+ * instead; then `parameters` stays unset (the block declares them).
1619
1754
  */
1620
- readonly instructions: string;
1755
+ readonly instructions: string | PromptRef;
1756
+ /** Settings blocks the agent reads, by range (see `Agent.settings`). */
1757
+ readonly settings?: readonly BlockRef[];
1758
+ /** A model-settings block, by range (see `Agent.modelSettings`). */
1759
+ readonly modelSettings?: BlockRef;
1621
1760
  /**
1622
1761
  * Required capabilities the agent needs from a `ModelProvider`.
1623
1762
  * Typically one entry: `[{ needs: [{ feature: 'tool-use' }] }]`
@@ -1812,6 +1951,11 @@ export interface ToolCompletedEvent {
1812
1951
  readonly invocationId: string;
1813
1952
  readonly output: unknown;
1814
1953
  readonly durationMs: number;
1954
+ /**
1955
+ * In a replay turn: whether the tool ran (`live`), the past run's result
1956
+ * was used (`recorded`), or the call was refused (`refused`).
1957
+ */
1958
+ readonly replay?: "live" | "recorded" | "refused";
1815
1959
  }
1816
1960
  export interface ToolFailedEvent {
1817
1961
  readonly kind: "tool.failed";
@@ -3772,6 +3916,130 @@ export interface ClientOptions {
3772
3916
  /** Overridable fetch impl for testing. Defaults to global `fetch`. */
3773
3917
  readonly fetch?: typeof fetch;
3774
3918
  }
3919
+ /** The verdict of a judgment. */
3920
+ export type Verdict = "yes" | "no";
3921
+ /**
3922
+ * Where a judge class applies: the whole tenant, one project, or one
3923
+ * agent in a project. Matches `@kindgi/api/openapi.json#JudgeClassScope`.
3924
+ */
3925
+ export type JudgeClassScope = {
3926
+ readonly kind: "tenant";
3927
+ } | {
3928
+ readonly kind: "project";
3929
+ readonly projectId: string;
3930
+ } | {
3931
+ readonly kind: "agent";
3932
+ readonly projectId: string;
3933
+ readonly agentId: string;
3934
+ };
3935
+ /** Matches `@kindgi/api/openapi.json#JudgeClass`. */
3936
+ export interface JudgeClass {
3937
+ readonly id: string;
3938
+ readonly tenantId: string;
3939
+ readonly scope: JudgeClassScope;
3940
+ /** The deployment's own word for the class: "expert", "user", "arbitrator". */
3941
+ readonly name: string;
3942
+ /** How much a judgment of this class counts, relative to the others (≥ 0). */
3943
+ readonly weight: number;
3944
+ readonly description?: string;
3945
+ readonly createdAt: Timestamp;
3946
+ readonly updatedAt: Timestamp;
3947
+ /** Set when the class was retired. */
3948
+ readonly unregisteredAt?: Timestamp;
3949
+ }
3950
+ /** Input for `POST /v1/judge-classes` per `#CreateJudgeClassBody`. */
3951
+ export interface CreateJudgeClassInput {
3952
+ readonly scope: JudgeClassScope;
3953
+ readonly name: string;
3954
+ readonly weight: number;
3955
+ readonly description?: string;
3956
+ }
3957
+ /** Input for `PATCH /v1/judge-classes/{judgeClassId}` per `#UpdateJudgeClassBody`. */
3958
+ export interface UpdateJudgeClassInput {
3959
+ readonly weight?: number;
3960
+ readonly description?: string;
3961
+ }
3962
+ /** What a judged run ran: an agent at a version, or a flow at a version. */
3963
+ export interface JudgedSubject {
3964
+ readonly kind: "agent" | "flow";
3965
+ readonly id: string;
3966
+ readonly version: string;
3967
+ }
3968
+ /** The judged item of a run's output. */
3969
+ export interface JudgedItem {
3970
+ /** Your stable id for the item, e.g. a matched case's id. */
3971
+ readonly key: string;
3972
+ /** Where the item is in the run's output, as a JSON Pointer (RFC 6901), e.g. `/matches/2`. */
3973
+ readonly pointer?: string;
3974
+ /** The item's position in a ranked list (0 = first). */
3975
+ readonly rank?: number;
3976
+ }
3977
+ /** Who asserted a judgment: the authenticated caller, never a typed name. */
3978
+ export interface JudgmentAssertedBy {
3979
+ readonly kind: "user" | "service";
3980
+ readonly id: string;
3981
+ }
3982
+ /** Matches `@kindgi/api/openapi.json#Judgment`. */
3983
+ export interface Judgment {
3984
+ readonly id: string;
3985
+ readonly tenantId: string;
3986
+ readonly projectId: string;
3987
+ readonly runId: string;
3988
+ readonly subject: JudgedSubject;
3989
+ readonly item: JudgedItem;
3990
+ readonly verdict: Verdict;
3991
+ readonly reason?: string;
3992
+ /** Absent when the judgment is unclassified (it counts with weight 1). */
3993
+ readonly judgeClassId?: string;
3994
+ readonly assertedBy: JudgmentAssertedBy;
3995
+ /** An app's opaque id for its end user who judged, when it judged on their behalf. */
3996
+ readonly participantId?: string;
3997
+ readonly createdAt: Timestamp;
3998
+ /** Set when the judgment was removed or superseded. */
3999
+ readonly unregisteredAt?: Timestamp;
4000
+ /** The judgment that replaced this one. */
4001
+ readonly supersededBy?: string;
4002
+ }
4003
+ /**
4004
+ * What a judged agent turn read besides its input, captured when it was
4005
+ * first judged: the conversation before it and what its retrievals returned.
4006
+ */
4007
+ export interface JudgedRunContext {
4008
+ /** The conversation's messages before the turn, oldest first (at most the last 200). */
4009
+ readonly history?: readonly unknown[];
4010
+ /** Whether older messages were left out of `history`. */
4011
+ readonly historyTruncated?: boolean;
4012
+ /** What the turn's retrievals returned. */
4013
+ readonly retrieved?: unknown;
4014
+ }
4015
+ /** The stored copy of a judged run's input and output. */
4016
+ export interface JudgedRunCopy {
4017
+ readonly runId: string;
4018
+ readonly subject: JudgedSubject;
4019
+ readonly input: unknown;
4020
+ /** Absent for runs judged before context was captured, and for flow runs. */
4021
+ readonly context?: JudgedRunContext;
4022
+ readonly output: unknown;
4023
+ readonly capturedAt: Timestamp;
4024
+ }
4025
+ /** Matches `@kindgi/api/openapi.json#JudgmentWithCopies`. */
4026
+ export interface JudgmentWithCopies extends Judgment {
4027
+ /** The run's input and output as they were when it was first judged. */
4028
+ readonly run: JudgedRunCopy;
4029
+ /** The judged item's value, when the judgment pointed at it. */
4030
+ readonly itemValue?: unknown;
4031
+ }
4032
+ /** Input for `POST /v1/judgments` per `#CreateJudgmentBody`. */
4033
+ export interface CreateJudgmentInput {
4034
+ readonly runId: string;
4035
+ readonly item: JudgedItem;
4036
+ readonly verdict: Verdict;
4037
+ readonly reason?: string;
4038
+ /** Optional. When given it must exist and apply to the run. */
4039
+ readonly judgeClassId?: string;
4040
+ /** Your opaque id for the end user who judged, when judging on their behalf. */
4041
+ readonly participantId?: string;
4042
+ }
3775
4043
  /** Signature the resource clients call into. */
3776
4044
  export interface Transport {
3777
4045
  /** Base URL, no trailing slash. */
@@ -3970,6 +4238,19 @@ declare namespace Schemas {
3970
4238
  version: string;
3971
4239
  conversationId: string;
3972
4240
  };
4241
+ /**
4242
+ * Agents and tools a flow runs at other exact versions than the flow version's pins ("this flow, with `acme.scorer` at 0.4.0"), without publishing a new flow version: a comparison's flow candidate (`versions` on the start body and the run's `comparison`), and the flow runs that replay it (a run's `versions`). Each id must be an agent or tool the flow uses.
4243
+ */
4244
+ export type FlowVersionOverrides = Partial<{
4245
+ /**
4246
+ * Tool id → exact version.
4247
+ */
4248
+ tools: Record<string, string>;
4249
+ /**
4250
+ * Agent id → exact version, for agent nodes with or without a version of their own.
4251
+ */
4252
+ agents: Record<string, string>;
4253
+ }>;
3973
4254
  export type Run = {
3974
4255
  /**
3975
4256
  * RunId.
@@ -4001,6 +4282,15 @@ declare namespace Schemas {
4001
4282
  */
4002
4283
  parentNodeId?: string;
4003
4284
  agent?: RunAgent;
4285
+ /**
4286
+ * Set on a replay run (an eval run re-running a past run): the run it replays.
4287
+ */
4288
+ replayOf?: string;
4289
+ /**
4290
+ * Set on a replay run: the eval run that started it.
4291
+ */
4292
+ evalRunId?: string;
4293
+ versions?: FlowVersionOverrides;
4004
4294
  /**
4005
4295
  * Only in the response to `POST /v1/runs`, when the deployment issues public run tokens: a read-only token for this run (and its descendants) to hand to a browser, for `GET /v1/runs/{runId}/progress` and its stream.
4006
4296
  */
@@ -4110,6 +4400,15 @@ declare namespace Schemas {
4110
4400
  */
4111
4401
  parentNodeId?: string;
4112
4402
  agent?: RunAgent;
4403
+ /**
4404
+ * Set on a replay run (an eval run re-running a past run): the run it replays.
4405
+ */
4406
+ replayOf?: string;
4407
+ /**
4408
+ * Set on a replay run: the eval run that started it.
4409
+ */
4410
+ evalRunId?: string;
4411
+ versions?: FlowVersionOverrides;
4113
4412
  /**
4114
4413
  * Only in the response to `POST /v1/runs`, when the deployment issues public run tokens: a read-only token for this run (and its descendants) to hand to a browser, for `GET /v1/runs/{runId}/progress` and its stream.
4115
4414
  */
@@ -4539,158 +4838,632 @@ declare namespace Schemas {
4539
4838
  nextCursor?: string;
4540
4839
  hasMore: boolean;
4541
4840
  };
4542
- export type PromptParameter = {
4841
+ /**
4842
+ * Where a judge class applies: the whole tenant, one project, or one agent in a project.
4843
+ */
4844
+ export type JudgeClassScope = {
4845
+ kind: "tenant";
4846
+ } | {
4847
+ kind: "project";
4848
+ projectId: string;
4849
+ } | {
4850
+ kind: "agent";
4851
+ projectId: string;
4852
+ agentId: string;
4853
+ };
4854
+ export type JudgeClass = {
4855
+ id: string;
4856
+ tenantId: string;
4857
+ scope: JudgeClassScope;
4858
+ /**
4859
+ * The deployment's own word for the class: "expert", "user", "arbitrator".
4860
+ */
4543
4861
  name: string;
4862
+ /**
4863
+ * How much a judgment of this class counts, relative to the others.
4864
+ */
4865
+ weight: number;
4544
4866
  description?: string;
4545
- type: "string" | "number" | "boolean" | "date";
4546
- required?: boolean;
4867
+ createdAt: string;
4868
+ updatedAt: string;
4547
4869
  /**
4548
- * Default value used when the caller omits this parameter.
4870
+ * Set when the class was retired.
4549
4871
  */
4550
- default?: unknown;
4872
+ unregisteredAt?: string;
4551
4873
  };
4552
- export type RetrievalIntent = {
4553
- types: Array<string>;
4554
- scope: "same-conversation" | "same-project" | "tenant";
4555
- limit?: number;
4556
- mode?: "keyword" | "semantic" | "both";
4874
+ export type JudgeClassCollectionPage = {
4875
+ data: Array<JudgeClass>;
4876
+ /**
4877
+ * Opaque cursor. Treat as opaque on the client.
4878
+ */
4879
+ nextCursor?: string;
4880
+ hasMore: boolean;
4557
4881
  };
4558
- export type ConversationPolicy = Partial<{
4559
- historyLimit: number;
4560
- autoCloseAfterInactiveSeconds: number;
4561
- hitlAfterTurns: number;
4562
- }>;
4563
- export type TurnBudget = Partial<{
4564
- maxSteps: number;
4565
- maxCostUsd: number;
4566
- maxWallMs: number;
4882
+ export type CreateJudgeClassBody = {
4883
+ scope: JudgeClassScope;
4884
+ name: string;
4885
+ weight: number;
4886
+ description?: string;
4887
+ };
4888
+ export type UpdateJudgeClassBody = Partial<{
4889
+ weight: number;
4890
+ description: string;
4567
4891
  }>;
4892
+ export type UnregisterJudgeClassResult = {
4893
+ judgeClassId: string;
4894
+ unregistered: true;
4895
+ };
4568
4896
  /**
4569
- * Capability declaration — see `@kindgi/capabilities`. Additional properties are permitted so new capability kinds do not require a wire change.
4570
- */
4571
- export type Capability = Record<string, unknown>;
4572
- /**
4573
- * Typed tool reference. `version` is a semver **range** (npm-style: `1.2.3` exact pin, `^1.2.3` compatible-updates, `~1.2.3` patch-updates-only, `>=1.0.0 <2.0.0` explicit range). Dispatch resolves the range to a concrete active version via `semver.maxSatisfying` at run start. No implicit "latest" — every tool ref names both id and range.
4897
+ * What a judged run ran: an agent at a version, or a flow at a version.
4574
4898
  */
4575
- export type ToolRef = {
4899
+ export type JudgedSubject = {
4900
+ kind: "agent" | "flow";
4576
4901
  id: string;
4577
4902
  version: string;
4578
4903
  };
4579
- export type AgentOutputSpec = {
4904
+ /**
4905
+ * The judged item of a run's output.
4906
+ */
4907
+ export type JudgedItem = {
4580
4908
  /**
4581
- * JSON Schema (draft 2020-12) the final answer must match.
4909
+ * The caller's stable id for the item.
4582
4910
  */
4583
- schema: Record<string, unknown>;
4911
+ key: string;
4584
4912
  /**
4585
- * A name for the output, shown to the model and in errors. Default `output`.
4913
+ * Where the item is in the run's output, as a JSON Pointer (RFC 6901), e.g. `/matches/2`. `""` is the whole output.
4586
4914
  */
4587
- name?: string;
4915
+ pointer?: string;
4588
4916
  /**
4589
- * How many times the model is asked to repair an invalid answer. Default 1.
4917
+ * The item's position in a ranked list (0 = first).
4590
4918
  */
4591
- maxRepairs?: number;
4919
+ rank?: number;
4592
4920
  };
4593
- export type ToolErrorsSpec = Partial<{
4594
- /**
4595
- * Failed calls sent back to the model per turn. Default 1.
4596
- */
4597
- maxRetries: number;
4921
+ /**
4922
+ * Who asserted a judgment: the authenticated caller, never a typed name.
4923
+ */
4924
+ export type JudgmentAssertedBy = {
4925
+ kind: "user" | "service";
4598
4926
  /**
4599
- * Which failures are sent back: arguments that don't fit the input schema (`invalid-arguments`), a tool the agent doesn't have (`unknown-tool`), a tool that ran and failed (`tool-error`). Default `invalid-arguments`, `unknown-tool`.
4927
+ * A user id, or for a service token its token or session id.
4600
4928
  */
4601
- retryOn: Array<"invalid-arguments" | "unknown-tool" | "tool-error">;
4602
- }>;
4603
- export type Agent = {
4929
+ id: string;
4930
+ };
4931
+ export type Judgment = {
4932
+ id: string;
4933
+ tenantId: string;
4934
+ projectId: string;
4935
+ runId: string;
4936
+ subject: JudgedSubject;
4937
+ item: JudgedItem;
4938
+ verdict: "yes" | "no";
4939
+ reason?: string;
4604
4940
  /**
4605
- * AgentId — dotted namespace (e.g. `acme.drafting`).
4941
+ * The judge class the judgment is recorded under. Absent when unclassified (counts with weight 1).
4606
4942
  */
4607
- id: string;
4943
+ judgeClassId?: string;
4944
+ assertedBy: JudgmentAssertedBy;
4608
4945
  /**
4609
- * Semver.
4946
+ * The app's opaque id for its end user who judged, when an app judged on their behalf.
4610
4947
  */
4611
- version: string;
4612
- name: string;
4613
- description?: string;
4614
- instructions: string;
4615
- parameters?: Array<PromptParameter>;
4616
- capabilities: Array<Capability>;
4617
- tools: Array<ToolRef>;
4618
- retrieval: Array<RetrievalIntent>;
4619
- guardrails: Array<string>;
4948
+ participantId?: string;
4949
+ createdAt: string;
4620
4950
  /**
4621
- * Soft hint — the router prefers this provider by id (e.g. `anthropic`) when at least one of its models satisfies `capabilities.needs` + tenant policy. Combine with `preferredModel` to pin the exact (provider, model) tuple. Falls back to capability-based ranking when the pinned provider is unregistered or filtered out.
4951
+ * Set when the judgment was removed or superseded.
4622
4952
  */
4623
- preferredProvider?: string;
4953
+ unregisteredAt?: string;
4624
4954
  /**
4625
- * Soft hint at the model level (`ModelInfo.name`, e.g. `claude-sonnet-4-6`). Combined with `preferredProvider`: both set → promote the exact tuple; only `preferredModel` → promote any provider exposing that model; only `preferredProvider` → promote every model of that provider.
4955
+ * The judgment that replaced this one.
4626
4956
  */
4627
- preferredModel?: string;
4628
- conversationPolicy?: ConversationPolicy;
4629
- budget?: TurnBudget;
4630
- tags?: Array<string>;
4631
- output?: AgentOutputSpec;
4632
- toolErrors?: ToolErrorsSpec;
4957
+ supersededBy?: string;
4633
4958
  };
4634
4959
  /**
4635
- * Full `defineAgent` spec. Validated server-side via `@kindgi/agents.defineAgent` — validation failures return `400 validation-failed` with the issue list under `details.issues`.
4960
+ * What a judged run needs besides its input to be replayed, captured when it was first judged. For an agent turn: the conversation before it, what its retrievals returned, and the decision at its session approval gate. For a flow run: its tool calls with their results.
4636
4961
  */
4637
- export type PublishAgentBody = {
4638
- id: string;
4639
- version: string;
4640
- name: string;
4641
- description?: string;
4962
+ export type JudgedRunContext = Partial<{
4642
4963
  /**
4643
- * Project this belongs to (its content scope). Required: missing, or not a project in the caller's tenant → `400 bad-input`.
4964
+ * The conversation's messages before the turn, oldest first (at most the last 200).
4644
4965
  */
4645
- projectId: string;
4646
- instructions: string;
4647
- parameters?: Array<PromptParameter>;
4648
- capabilities: Array<Capability>;
4649
- tools: Array<ToolRef>;
4650
- retrieval: Array<RetrievalIntent>;
4651
- guardrails: Array<string>;
4966
+ history: Array<unknown>;
4652
4967
  /**
4653
- * Soft hint — the router prefers this provider by id (e.g. `anthropic`) when at least one of its models satisfies `capabilities.needs` + tenant policy. Combine with `preferredModel` to pin the exact (provider, model) tuple.
4968
+ * Whether older messages were left out of `history`.
4654
4969
  */
4655
- preferredProvider?: string;
4970
+ historyTruncated: boolean;
4656
4971
  /**
4657
- * Soft hint at the model level (`ModelInfo.name`). Combined with `preferredProvider` to pin an exact tuple; alone to select a model across every provider that exposes it.
4972
+ * What the turn's retrievals returned.
4658
4973
  */
4659
- preferredModel?: string;
4660
- conversationPolicy?: ConversationPolicy;
4661
- budget?: TurnBudget;
4662
- tags?: Array<string>;
4663
- output?: AgentOutputSpec;
4664
- toolErrors?: ToolErrorsSpec;
4665
- };
4666
- export type PublishAgentResult = {
4667
- agentId: string;
4668
- version: string;
4669
- };
4670
- export type UnregisterAgentResult = {
4671
- agentId: string;
4672
- version: string;
4673
- unregistered: true;
4674
- };
4675
- export type ReinstateAgentVersionResult = {
4676
- agentId: string;
4677
- version: string;
4974
+ retrieved: unknown;
4678
4975
  /**
4679
- * `true` when this call un-tombstoned the version; `false` when it was already active (idempotent no-op).
4976
+ * The reviewer's decision at the turn's session approval gate, when the turn waited on one. A replay of the turn follows it.
4680
4977
  */
4681
- wasTombstoned: boolean;
4682
- };
4683
- export type AgentCollectionPage = {
4684
- data: Array<Agent>;
4978
+ sessionApproval: {
4979
+ approved: boolean;
4980
+ /**
4981
+ * The reviewer's reason for a rejection.
4982
+ */
4983
+ rationale?: string;
4984
+ };
4685
4985
  /**
4686
- * Opaque cursor for the next page. Absent when `hasMore: false`.
4687
- */
4688
- nextCursor?: string;
4689
- hasMore: boolean;
4690
- };
4691
- export type FlowNode = {
4692
- id: string;
4693
- kind: "tool" | "agent" | "loop" | "fanout" | "subgraph";
4986
+ * For a flow run: what it did, kept at its first judgment so it can be replayed. Every tool call it made with its result (at its tool nodes, in its agent steps' turns and in its sub-flows), at most 500, and its agent steps.
4987
+ */
4988
+ flow: {
4989
+ calls: Array<{
4990
+ /**
4991
+ * The run that made it: the flow run, a sub-flow's, or an agent step's turn.
4992
+ */
4993
+ runId: string;
4994
+ /**
4995
+ * The tool node that made it, or the agent step whose turn did.
4996
+ */
4997
+ nodeId?: string;
4998
+ /**
4999
+ * The loop iteration, in a loop body.
5000
+ */
5001
+ scope?: string;
5002
+ toolId: string;
5003
+ arguments?: unknown;
5004
+ result?: unknown;
5005
+ }>;
5006
+ steps: Array<{
5007
+ runId: string;
5008
+ nodeId?: string;
5009
+ scope?: string;
5010
+ agentId: string;
5011
+ agentVersion: string;
5012
+ /**
5013
+ * What the step's turn retrieved.
5014
+ */
5015
+ retrieved?: unknown;
5016
+ }>;
5017
+ /**
5018
+ * More calls were made than were kept.
5019
+ */
5020
+ truncated?: boolean;
5021
+ };
5022
+ }>;
5023
+ /**
5024
+ * The stored copy of a judged run's input and output, taken when it was first judged.
5025
+ */
5026
+ export type JudgedRunCopy = {
5027
+ runId: string;
5028
+ subject: JudgedSubject;
5029
+ input: unknown;
5030
+ context?: JudgedRunContext;
5031
+ output: unknown;
5032
+ capturedAt: string;
5033
+ };
5034
+ /**
5035
+ * A judgment with the stored copies of what was judged.
5036
+ */
5037
+ export type JudgmentWithCopies = Judgment & {
5038
+ run: JudgedRunCopy;
5039
+ /**
5040
+ * The judged item's value, when the judgment pointed at it.
5041
+ */
5042
+ itemValue?: unknown;
5043
+ };
5044
+ export type JudgmentCollectionPage = {
5045
+ data: Array<Judgment>;
5046
+ /**
5047
+ * Opaque cursor. Treat as opaque on the client.
5048
+ */
5049
+ nextCursor?: string;
5050
+ hasMore: boolean;
5051
+ };
5052
+ export type CreateJudgmentBody = {
5053
+ runId: string;
5054
+ item: JudgedItem;
5055
+ verdict: "yes" | "no";
5056
+ reason?: string;
5057
+ /**
5058
+ * Optional. When given it must exist and apply to the run.
5059
+ */
5060
+ judgeClassId?: string;
5061
+ /**
5062
+ * An app's opaque id for its end user, when judging on their behalf.
5063
+ */
5064
+ participantId?: string;
5065
+ };
5066
+ export type UnregisterJudgmentResult = {
5067
+ judgmentId: string;
5068
+ unregistered: true;
5069
+ };
5070
+ /**
5071
+ * The judgments of one item of a case's output, summed up.
5072
+ */
5073
+ export type JudgedItemSummary = {
5074
+ key: string;
5075
+ pointer?: string;
5076
+ rank?: number;
5077
+ /**
5078
+ * How many judgments said yes.
5079
+ */
5080
+ yes: number;
5081
+ /**
5082
+ * How many judgments said no.
5083
+ */
5084
+ no: number;
5085
+ /**
5086
+ * The weight behind "yes" (an unclassified judgment counts 1).
5087
+ */
5088
+ yesWeight: number;
5089
+ /**
5090
+ * The weight behind all judgments of the item.
5091
+ */
5092
+ totalWeight: number;
5093
+ /**
5094
+ * The reasons given, newest first.
5095
+ */
5096
+ reasons: Array<{
5097
+ verdict: "yes" | "no";
5098
+ reason: string;
5099
+ }>;
5100
+ };
5101
+ /**
5102
+ * One case of a `judged` eval suite: a copy of a judged run with its items' judgments summed up.
5103
+ */
5104
+ export type JudgedEvalCase = {
5105
+ /**
5106
+ * The judged run's id.
5107
+ */
5108
+ caseId: string;
5109
+ subject: JudgedSubject;
5110
+ input: unknown;
5111
+ context?: JudgedRunContext;
5112
+ output: unknown;
5113
+ items: Array<JudgedItemSummary>;
5114
+ };
5115
+ export type JudgedEvalCaseCollectionPage = {
5116
+ data: Array<JudgedEvalCase>;
5117
+ /**
5118
+ * Opaque cursor. Treat as opaque on the client.
5119
+ */
5120
+ nextCursor?: string;
5121
+ hasMore: boolean;
5122
+ };
5123
+ /**
5124
+ * Name the agent (`agentId`) or the flow (`flowId`) whose judged runs to use.
5125
+ */
5126
+ export type BuildJudgedSuiteBody = {
5127
+ /**
5128
+ * Semver version to publish, e.g. `1.0.0`.
5129
+ */
5130
+ version: string;
5131
+ projectId: string;
5132
+ agentId?: string;
5133
+ /**
5134
+ * Needs `agentId`.
5135
+ */
5136
+ agentVersion?: string;
5137
+ flowId?: string;
5138
+ /**
5139
+ * Runs first judged at or after this time.
5140
+ */
5141
+ since?: string;
5142
+ /**
5143
+ * Runs first judged before this time.
5144
+ */
5145
+ until?: string;
5146
+ /**
5147
+ * Count only judgments recorded under these judge classes.
5148
+ */
5149
+ judgeClassIds?: Array<string>;
5150
+ /**
5151
+ * Leave out runs with fewer counted judgments. Default 1.
5152
+ */
5153
+ minJudgments?: number;
5154
+ description?: string;
5155
+ };
5156
+ export type BuildJudgedSuiteResult = {
5157
+ suiteId: string;
5158
+ version: string;
5159
+ kind: "judged";
5160
+ caseCount: number;
5161
+ /**
5162
+ * Whether more judged runs matched than the 1000 cases a set holds.
5163
+ */
5164
+ truncated: boolean;
5165
+ };
5166
+ export type PromptParameter = {
5167
+ name: string;
5168
+ description?: string;
5169
+ type: "string" | "number" | "boolean" | "date";
5170
+ required?: boolean;
5171
+ /**
5172
+ * Default value used when the caller omits this parameter.
5173
+ */
5174
+ default?: unknown;
5175
+ };
5176
+ export type RetrievalIntent = {
5177
+ types: Array<string>;
5178
+ scope: "same-conversation" | "same-project" | "tenant";
5179
+ limit?: number;
5180
+ mode?: "keyword" | "semantic" | "both";
5181
+ };
5182
+ export type ConversationPolicy = Partial<{
5183
+ historyLimit: number;
5184
+ autoCloseAfterInactiveSeconds: number;
5185
+ hitlAfterTurns: number;
5186
+ }>;
5187
+ export type TurnBudget = Partial<{
5188
+ maxSteps: number;
5189
+ maxCostUsd: number;
5190
+ maxWallMs: number;
5191
+ }>;
5192
+ /**
5193
+ * Capability declaration — see `@kindgi/capabilities`. Additional properties are permitted so new capability kinds do not require a wire change.
5194
+ */
5195
+ export type Capability = Record<string, unknown>;
5196
+ /**
5197
+ * Typed tool reference. `version` is a semver **range** (npm-style: `1.2.3` exact pin, `^1.2.3` compatible-updates, `~1.2.3` patch-updates-only, `>=1.0.0 <2.0.0` explicit range). Dispatch resolves the range to a concrete active version via `semver.maxSatisfying` at run start. No implicit "latest" — every tool ref names both id and range.
5198
+ */
5199
+ export type ToolRef = {
5200
+ id: string;
5201
+ version: string;
5202
+ };
5203
+ export type AgentOutputSpec = {
5204
+ /**
5205
+ * JSON Schema (draft 2020-12) the final answer must match.
5206
+ */
5207
+ schema: Record<string, unknown>;
5208
+ /**
5209
+ * A name for the output, shown to the model and in errors. Default `output`.
5210
+ */
5211
+ name?: string;
5212
+ /**
5213
+ * How many times the model is asked to repair an invalid answer. Default 1.
5214
+ */
5215
+ maxRepairs?: number;
5216
+ };
5217
+ export type ToolErrorsSpec = Partial<{
5218
+ /**
5219
+ * Failed calls sent back to the model per turn. Default 1.
5220
+ */
5221
+ maxRetries: number;
5222
+ /**
5223
+ * Which failures are sent back: arguments that don't fit the input schema (`invalid-arguments`), a tool the agent doesn't have (`unknown-tool`), a tool that ran and failed (`tool-error`). Default `invalid-arguments`, `unknown-tool`.
5224
+ */
5225
+ retryOn: Array<"invalid-arguments" | "unknown-tool" | "tool-error">;
5226
+ }>;
5227
+ /**
5228
+ * A prompt block an agent's instructions come from, by id and semver range.
5229
+ */
5230
+ export type PromptRef = {
5231
+ /**
5232
+ * The prompt block id.
5233
+ */
5234
+ prompt: string;
5235
+ /**
5236
+ * A semver range (`^1.0.0`, `1.2.0`).
5237
+ */
5238
+ version: string;
5239
+ };
5240
+ /**
5241
+ * A settings block an agent reads, by id and semver range.
5242
+ */
5243
+ export type BlockRef = {
5244
+ /**
5245
+ * The settings block id.
5246
+ */
5247
+ id: string;
5248
+ /**
5249
+ * A semver range (`^1.0.0`, `1.2.0`).
5250
+ */
5251
+ version: string;
5252
+ };
5253
+ /**
5254
+ * The exact block versions an agent version runs: its lockfile. Set by the runtime when the version is published, never in the publish body: each tool range resolves once to the version every run of that agent version uses, so a new tool version reaches the agent only through a new agent version. Absent on a version published before pins existed (its ranges resolve per run).
5255
+ */
5256
+ export type AgentPins = {
5257
+ /**
5258
+ * Tool id → exact version.
5259
+ */
5260
+ tools: Record<string, string>;
5261
+ /**
5262
+ * Prompt block id → exact version.
5263
+ */
5264
+ prompts: Record<string, string>;
5265
+ /**
5266
+ * Settings block id → exact version.
5267
+ */
5268
+ settings: Record<string, string>;
5269
+ };
5270
+ /**
5271
+ * Set by the runtime on an agent or flow version a deploy registered in place of the definition's version, which was registered already with other pins or content (versions never change). Never in the publish body.
5272
+ */
5273
+ export type VersionDerivation = {
5274
+ /**
5275
+ * The version the definition names.
5276
+ */
5277
+ version: string;
5278
+ /**
5279
+ * `pins-changed`: a block it uses has a new version; `unpinned`: the definition's version was published before pins existed; `version-taken`: the definition's version holds another definition; `edited`: derived from `version` with some data-block pins swapped (`POST /v1/agents/{agentId}/versions`).
5280
+ */
5281
+ reason: "pins-changed" | "unpinned" | "version-taken" | "edited";
5282
+ /**
5283
+ * For `edited`: a short label for the version.
5284
+ */
5285
+ label?: string;
5286
+ /**
5287
+ * For `edited`: who derived it (`user:<id>`).
5288
+ */
5289
+ by?: string;
5290
+ };
5291
+ export type Agent = {
5292
+ /**
5293
+ * AgentId — dotted namespace (e.g. `acme.drafting`).
5294
+ */
5295
+ id: string;
5296
+ /**
5297
+ * Semver.
5298
+ */
5299
+ version: string;
5300
+ name: string;
5301
+ description?: string;
5302
+ /**
5303
+ * The system prompt (a Liquid template), or a prompt block by range whose template and parameters are used instead (pinned at publish, `pins.prompts`).
5304
+ */
5305
+ instructions: string | PromptRef;
5306
+ parameters?: Array<PromptParameter>;
5307
+ /**
5308
+ * Settings blocks the agent reads, by range: tools read them as `ToolContext.settings['<id>']`, templates as `settings["<id>"]`. Pinned at publish (`pins.settings`).
5309
+ */
5310
+ settings?: Array<BlockRef>;
5311
+ modelSettings?: BlockRef;
5312
+ capabilities: Array<Capability>;
5313
+ tools: Array<ToolRef>;
5314
+ retrieval: Array<RetrievalIntent>;
5315
+ guardrails: Array<string>;
5316
+ /**
5317
+ * Soft hint — the router prefers this provider by id (e.g. `anthropic`) when at least one of its models satisfies `capabilities.needs` + tenant policy. Combine with `preferredModel` to pin the exact (provider, model) tuple. Falls back to capability-based ranking when the pinned provider is unregistered or filtered out.
5318
+ */
5319
+ preferredProvider?: string;
5320
+ /**
5321
+ * Soft hint at the model level (`ModelInfo.name`, e.g. `claude-sonnet-4-6`). Combined with `preferredProvider`: both set → promote the exact tuple; only `preferredModel` → promote any provider exposing that model; only `preferredProvider` → promote every model of that provider.
5322
+ */
5323
+ preferredModel?: string;
5324
+ conversationPolicy?: ConversationPolicy;
5325
+ budget?: TurnBudget;
5326
+ tags?: Array<string>;
5327
+ output?: AgentOutputSpec;
5328
+ toolErrors?: ToolErrorsSpec;
5329
+ pins?: AgentPins;
5330
+ derivedFrom?: VersionDerivation;
5331
+ /**
5332
+ * Present only on an unregistered version (`GET …/versions/{version}` reads those too). Unregister stops a version being chosen, not the pins that hold it: a new run naming it is refused, while a resumed run and a published version that pins it still run it.
5333
+ */
5334
+ unregisteredAt?: string;
5335
+ /**
5336
+ * Set by the runtime with `pins`: `sha256:<hex>` of the pins' canonical JSON (sorted keys, no whitespace). Two agent versions with the same digest run the same blocks.
5337
+ */
5338
+ pinsDigest?: string;
5339
+ };
5340
+ /**
5341
+ * One pin that differs between two versions of an agent or a flow.
5342
+ */
5343
+ export type PinChange = {
5344
+ kind: "tool" | "prompt" | "setting" | "agent";
5345
+ id: string;
5346
+ /**
5347
+ * The earlier version's pin; absent when it had none.
5348
+ */
5349
+ from?: string;
5350
+ /**
5351
+ * The later version's pin; absent when it has none.
5352
+ */
5353
+ to?: string;
5354
+ };
5355
+ /**
5356
+ * The data-block pins to swap, by block id → exact version. Only blocks the version already references; tool pins come from code.
5357
+ */
5358
+ export type AgentPinSwaps = Partial<{
5359
+ /**
5360
+ * Prompt block id → exact version.
5361
+ */
5362
+ prompts: Record<string, string>;
5363
+ /**
5364
+ * Settings block id → exact version.
5365
+ */
5366
+ settings: Record<string, string>;
5367
+ }>;
5368
+ /**
5369
+ * Derive a new agent version from a pinned one with some data-block pins swapped: an expert's edit reaching an agent with no code change.
5370
+ */
5371
+ export type DeriveAgentVersionBody = {
5372
+ /**
5373
+ * The version to derive from (it must be pinned).
5374
+ */
5375
+ from: string;
5376
+ pins: AgentPinSwaps;
5377
+ /**
5378
+ * A short label for the new version.
5379
+ */
5380
+ label?: string;
5381
+ /**
5382
+ * The agent's project, when the runtime doesn't record it on the version.
5383
+ */
5384
+ projectId?: string;
5385
+ };
5386
+ /**
5387
+ * The exact tool and agent versions a flow version runs: its lockfile. Set by the runtime when the version is published, never in the publish body: each tool the flow runs, and each agent it runs at no named version, resolves once to its latest version then, which every run of that flow version uses. Absent on a version published before pins existed (it binds the latest versions per run).
5388
+ */
5389
+ export type FlowPins = {
5390
+ /**
5391
+ * Tool id → exact version.
5392
+ */
5393
+ tools: Record<string, string>;
5394
+ /**
5395
+ * Agent id → exact version, for agent nodes that name no version.
5396
+ */
5397
+ agents: Record<string, string>;
5398
+ };
5399
+ /**
5400
+ * Full `defineAgent` spec. Validated server-side via `@kindgi/agents.defineAgent` — validation failures return `400 validation-failed` with the issue list under `details.issues`.
5401
+ */
5402
+ export type PublishAgentBody = {
5403
+ id: string;
5404
+ version: string;
5405
+ name: string;
5406
+ description?: string;
5407
+ /**
5408
+ * Project this belongs to (its content scope). Required: missing, or not a project in the caller's tenant → `400 bad-input`.
5409
+ */
5410
+ projectId: string;
5411
+ /**
5412
+ * The system prompt (a Liquid template), or a prompt block by range whose template and parameters are used instead (pinned at publish, `pins.prompts`).
5413
+ */
5414
+ instructions: string | PromptRef;
5415
+ parameters?: Array<PromptParameter>;
5416
+ /**
5417
+ * Settings blocks the agent reads, by range: tools read them as `ToolContext.settings['<id>']`, templates as `settings["<id>"]`. Pinned at publish (`pins.settings`).
5418
+ */
5419
+ settings?: Array<BlockRef>;
5420
+ modelSettings?: BlockRef;
5421
+ capabilities: Array<Capability>;
5422
+ tools: Array<ToolRef>;
5423
+ retrieval: Array<RetrievalIntent>;
5424
+ guardrails: Array<string>;
5425
+ /**
5426
+ * Soft hint — the router prefers this provider by id (e.g. `anthropic`) when at least one of its models satisfies `capabilities.needs` + tenant policy. Combine with `preferredModel` to pin the exact (provider, model) tuple.
5427
+ */
5428
+ preferredProvider?: string;
5429
+ /**
5430
+ * Soft hint at the model level (`ModelInfo.name`). Combined with `preferredProvider` to pin an exact tuple; alone to select a model across every provider that exposes it.
5431
+ */
5432
+ preferredModel?: string;
5433
+ conversationPolicy?: ConversationPolicy;
5434
+ budget?: TurnBudget;
5435
+ tags?: Array<string>;
5436
+ output?: AgentOutputSpec;
5437
+ toolErrors?: ToolErrorsSpec;
5438
+ };
5439
+ export type PublishAgentResult = {
5440
+ agentId: string;
5441
+ version: string;
5442
+ };
5443
+ export type UnregisterAgentResult = {
5444
+ agentId: string;
5445
+ version: string;
5446
+ unregistered: true;
5447
+ };
5448
+ export type ReinstateAgentVersionResult = {
5449
+ agentId: string;
5450
+ version: string;
5451
+ /**
5452
+ * `true` when this call un-tombstoned the version; `false` when it was already active (idempotent no-op).
5453
+ */
5454
+ wasTombstoned: boolean;
5455
+ };
5456
+ export type AgentCollectionPage = {
5457
+ data: Array<Agent>;
5458
+ /**
5459
+ * Opaque cursor for the next page. Absent when `hasMore: false`.
5460
+ */
5461
+ nextCursor?: string;
5462
+ hasMore: boolean;
5463
+ };
5464
+ export type FlowNode = {
5465
+ id: string;
5466
+ kind: "tool" | "agent" | "loop" | "fanout" | "subgraph";
4694
5467
  /**
4695
5468
  * Handler reference for leaf nodes.
4696
5469
  */
@@ -4731,6 +5504,16 @@ declare namespace Schemas {
4731
5504
  edges: Array<FlowEdge>;
4732
5505
  maxParallelism?: number;
4733
5506
  metadata?: Record<string, unknown>;
5507
+ pins?: FlowPins;
5508
+ /**
5509
+ * Set by the runtime with `pins`: `sha256:<hex>` of the pins' canonical JSON (sorted keys, no whitespace).
5510
+ */
5511
+ pinsDigest?: string;
5512
+ derivedFrom?: VersionDerivation;
5513
+ /**
5514
+ * Present only on an unregistered version (`GET …/versions/{version}` reads those too). Unregister stops a version being chosen, not the pins that hold it: a new run naming it is refused, while a resumed run and a published version that pins it still run it.
5515
+ */
5516
+ unregisteredAt?: string;
4734
5517
  };
4735
5518
  /**
4736
5519
  * Full flow definition. Validated server-side via `@kindgi/flow.loadFlow` — validation failures return `400 validation-failed` with the issue list under `details.issues`.
@@ -5741,6 +6524,10 @@ declare namespace Schemas {
5741
6524
  * A fallback serves a capability only when no other provider satisfies it (e.g. `kindgi dev`'s scripted `dev-echo`); an agent turn routed to one carries a `fallback-provider` warning. Absent = `false`.
5742
6525
  */
5743
6526
  fallback?: boolean;
6527
+ /**
6528
+ * Bookkeeping, such as who manages the provider; the router ignores labels. At most 32 keys; a key is 1-63 lowercase letters and digits, with `.`, `-`, `_` or `/` inside; a value is at most 256 characters. The convention key `kindgi.com/managed-by` names the manager (`kindgi-dev`, `kindgi-deploy:<environment>`). Out of bounds: `400 invalid-provider`, reason `invalid-labels`.
6529
+ */
6530
+ labels?: Record<string, string>;
5744
6531
  };
5745
6532
  export type ProviderCollectionPage = {
5746
6533
  data: Array<ProviderMetadata>;
@@ -6056,7 +6843,18 @@ declare namespace Schemas {
6056
6843
  tokens: CostTokenTotals;
6057
6844
  };
6058
6845
  export type CostAggregateResult = {
6846
+ /**
6847
+ * The most expensive groups first (`totalUsd` descending, ties by key), at most `limit`.
6848
+ */
6059
6849
  groups: Array<CostAggregateGroup>;
6850
+ /**
6851
+ * How many groups there were before the `limit` cap. Absent from a runtime before 0.1.5.
6852
+ */
6853
+ totalGroups?: number;
6854
+ /**
6855
+ * `true` when there were more groups than `limit`: `groups` holds the most expensive ones, and `totalUsd` / `totalRecords` / `tokens` still cover every record. Absent from a runtime before 0.1.5.
6856
+ */
6857
+ truncated?: boolean;
6060
6858
  totalUsd: number;
6061
6859
  totalRecords: number;
6062
6860
  tokens: CostTokenTotals;
@@ -6178,7 +6976,7 @@ declare namespace Schemas {
6178
6976
  version: string;
6179
6977
  wasTombstoned: boolean;
6180
6978
  };
6181
- export type EvalKind = "accuracy" | "pairwise" | "regression" | "human-review" | "benchmark" | "custom";
6979
+ export type EvalKind = "accuracy" | "pairwise" | "regression" | "human-review" | "benchmark" | "custom" | "judged";
6182
6980
  export type EvalSuite = {
6183
6981
  id: string;
6184
6982
  tenantId: string;
@@ -6207,23 +7005,96 @@ declare namespace Schemas {
6207
7005
  /**
6208
7006
  * Optional. When present, must match the caller tenant (server-derived from the token). Cross-tenant publish is rejected.
6209
7007
  */
6210
- tenantId?: string;
7008
+ tenantId?: string;
7009
+ version: string;
7010
+ kind: EvalKind;
7011
+ description?: string;
7012
+ spec: Record<string, unknown>;
7013
+ };
7014
+ export type PublishEvalSuiteResult = {
7015
+ suiteId: string;
7016
+ version: string;
7017
+ };
7018
+ export type UnregisterEvalSuiteResult = {
7019
+ suiteId: string;
7020
+ version: string;
7021
+ unregistered: true;
7022
+ };
7023
+ export type ReinstateEvalSuiteVersionResult = {
7024
+ suiteId: string;
7025
+ version: string;
7026
+ wasTombstoned: boolean;
7027
+ };
7028
+ /**
7029
+ * `prompt`: a Liquid template an agent renders as its instructions. `settings`: a JSON object tools and templates read.
7030
+ */
7031
+ export type BlockKind = "prompt" | "settings";
7032
+ export type PromptBlockContent = {
7033
+ /**
7034
+ * Liquid, rendered as an agent's instructions are (same parameters and auto-injected variables).
7035
+ */
7036
+ template: string;
7037
+ parameters?: Array<PromptParameter>;
7038
+ };
7039
+ export type SettingsBlockContent = {
7040
+ /**
7041
+ * What tools read (`ToolContext.settings[<block id>]`) and templates read (`settings.<block id>.<key>`).
7042
+ */
7043
+ values: Record<string, unknown>;
7044
+ /**
7045
+ * JSON Schema (draft 2020-12) the values must satisfy. A later version's values must satisfy the latest version's schema too.
7046
+ */
7047
+ schema?: Record<string, unknown>;
7048
+ };
7049
+ /**
7050
+ * A data block version: a prompt or settings, versioned like a tool (immutable versions, soft unregister). An agent version pins the block versions it uses when it is published. Belongs to one project and is authorized through it.
7051
+ */
7052
+ export type Block = {
7053
+ /**
7054
+ * Dotted lowercase id (e.g. `acme.intake-prompt`).
7055
+ */
7056
+ id: string;
7057
+ version: string;
7058
+ kind: BlockKind;
7059
+ description?: string;
7060
+ /**
7061
+ * `PromptBlockContent` for a prompt, `SettingsBlockContent` for settings.
7062
+ */
7063
+ content: PromptBlockContent | SettingsBlockContent;
7064
+ projectId: string;
7065
+ publishedAt: string;
7066
+ /**
7067
+ * Present only on an unregistered version; agent versions that pin it still read it.
7068
+ */
7069
+ unregisteredAt?: string;
7070
+ };
7071
+ export type BlockCollectionPage = {
7072
+ data: Array<Block>;
7073
+ nextCursor?: string;
7074
+ hasMore: boolean;
7075
+ };
7076
+ export type PublishBlockBody = {
7077
+ /**
7078
+ * Project this belongs to (its content scope). Required: missing, or not a project in the caller's tenant → `400 bad-input`.
7079
+ */
7080
+ projectId: string;
7081
+ id: string;
6211
7082
  version: string;
6212
- kind: EvalKind;
7083
+ kind: BlockKind;
6213
7084
  description?: string;
6214
- spec: Record<string, unknown>;
7085
+ content: PromptBlockContent | SettingsBlockContent;
6215
7086
  };
6216
- export type PublishEvalSuiteResult = {
6217
- suiteId: string;
7087
+ export type PublishBlockResult = {
7088
+ blockId: string;
6218
7089
  version: string;
6219
7090
  };
6220
- export type UnregisterEvalSuiteResult = {
6221
- suiteId: string;
7091
+ export type UnregisterBlockResult = {
7092
+ blockId: string;
6222
7093
  version: string;
6223
7094
  unregistered: true;
6224
7095
  };
6225
- export type ReinstateEvalSuiteVersionResult = {
6226
- suiteId: string;
7096
+ export type ReinstateBlockResult = {
7097
+ blockId: string;
6227
7098
  version: string;
6228
7099
  wasTombstoned: boolean;
6229
7100
  };
@@ -6236,6 +7107,223 @@ declare namespace Schemas {
6236
7107
  flowId: string;
6237
7108
  version?: string;
6238
7109
  };
7110
+ /**
7111
+ * What a comparison compares the candidate against: `'recorded'` (each case's recorded output, what was judged), a version (`{ agentId, version }`, replayed under the same rules), or the version live in a scope (`{ live: { projectId?, segments? } }`). Only `'recorded'` runs today; the others are refused when the run starts.
7112
+ */
7113
+ export type EvalBaseline = "recorded" | {
7114
+ agentId: string;
7115
+ version: string;
7116
+ } | {
7117
+ live: Partial<{
7118
+ projectId: string;
7119
+ segments: Record<string, string>;
7120
+ }>;
7121
+ };
7122
+ /**
7123
+ * A comparison eval run's settings (a `judged` suite).
7124
+ */
7125
+ export type EvalComparison = {
7126
+ baseline: EvalBaseline;
7127
+ /**
7128
+ * Whether replayed reads use the past run's results when it has them (`recorded`), or run live.
7129
+ */
7130
+ reads: "recorded" | "live";
7131
+ repetitions: number;
7132
+ k: number;
7133
+ versions?: FlowVersionOverrides;
7134
+ };
7135
+ /**
7136
+ * One metric, the recorded runs beside the candidate. `null` where a side had no judged evidence; `n` / `weight` are the candidate's evidence (cases with judged items, and the judgment weight behind them), `baselineN` / `baselineWeight` the recorded side's.
7137
+ */
7138
+ export type ComparisonMetric = {
7139
+ baseline: number | null;
7140
+ candidate: number | null;
7141
+ delta: number | null;
7142
+ n: number;
7143
+ weight: number;
7144
+ baselineN: number;
7145
+ baselineWeight: number;
7146
+ direction: "higher";
7147
+ /**
7148
+ * `weightedPrecisionAtK`: the ranked items it looked at.
7149
+ */
7150
+ k?: number;
7151
+ /**
7152
+ * With more than one repetition: the candidate's max − min across them.
7153
+ */
7154
+ spread?: number;
7155
+ };
7156
+ /**
7157
+ * What ran on the cases: an agent version, or a flow version (with any versions it swapped in).
7158
+ */
7159
+ export type ComparisonCandidate = {
7160
+ kind: "agent";
7161
+ agentId: string;
7162
+ version: string;
7163
+ } | {
7164
+ kind: "flow";
7165
+ flowId: string;
7166
+ version: string;
7167
+ versions?: FlowVersionOverrides;
7168
+ };
7169
+ /**
7170
+ * What a comparison concluded: the candidate beside the recorded runs, the case counts, and the metrics. What a promotion gate reads.
7171
+ */
7172
+ export type JudgedComparisonSummary = {
7173
+ evalRunId: string;
7174
+ status: "completed" | "partial" | "failed";
7175
+ completedAt: string;
7176
+ suite: {
7177
+ id: string;
7178
+ version: string;
7179
+ };
7180
+ candidate: ComparisonCandidate;
7181
+ /**
7182
+ * What the candidate was compared with: `recorded` (the test set's recorded runs, with the versions that served them), or another version.
7183
+ */
7184
+ baseline: {
7185
+ kind: "recorded";
7186
+ versions: Array<{
7187
+ agentId?: string;
7188
+ flowId?: string;
7189
+ version: string;
7190
+ cases: number;
7191
+ }>;
7192
+ } | {
7193
+ kind: "version";
7194
+ agentId: string;
7195
+ version: string;
7196
+ via: "explicit" | "live";
7197
+ liveScope?: Record<string, unknown>;
7198
+ };
7199
+ /**
7200
+ * Where the test set's judgments came from.
7201
+ */
7202
+ scope: Partial<{
7203
+ projectId: string;
7204
+ }>;
7205
+ cases: number;
7206
+ /**
7207
+ * Cases where a read with no recording ran live under `reads: 'recorded'`.
7208
+ */
7209
+ diverged: number;
7210
+ /**
7211
+ * Tool calls refused across the cases (what the candidate would have done).
7212
+ */
7213
+ refusedWrites: number;
7214
+ /**
7215
+ * Cases none of whose repetitions ran.
7216
+ */
7217
+ errors: number;
7218
+ /**
7219
+ * Flow cases that stopped at a write the replay refused: no output to score, so they're left out of the metrics.
7220
+ */
7221
+ stopped: number;
7222
+ reads: "recorded" | "live";
7223
+ sampling: {
7224
+ /**
7225
+ * The models that answered the candidate's replays, and how many replays each.
7226
+ */
7227
+ models: Array<{
7228
+ providerId: string;
7229
+ model: string;
7230
+ runs: number;
7231
+ }>;
7232
+ };
7233
+ repetitions: number;
7234
+ metrics: {
7235
+ weightedYesShare: ComparisonMetric;
7236
+ judgedCoverage: ComparisonMetric;
7237
+ weightedPrecisionAtK: ComparisonMetric;
7238
+ };
7239
+ };
7240
+ /**
7241
+ * One case of a comparison: its replay runs, the scores, the items kept, dropped and new, the tool calls, and why it didn't run when it didn't.
7242
+ */
7243
+ export type ComparisonCaseResult = {
7244
+ caseId: string;
7245
+ /**
7246
+ * The candidate's replay runs, one per repetition.
7247
+ */
7248
+ runIds: Array<string>;
7249
+ /**
7250
+ * An output's score: Σ yesWeight and Σ totalWeight over its judged items, and over those among the first `k` ranked items.
7251
+ */
7252
+ baseline: {
7253
+ yesWeight: number;
7254
+ totalWeight: number;
7255
+ items: number;
7256
+ judgedItems: number;
7257
+ topK: {
7258
+ yesWeight: number;
7259
+ totalWeight: number;
7260
+ };
7261
+ };
7262
+ /**
7263
+ * One per repetition that ran.
7264
+ */
7265
+ candidate: Array<{
7266
+ yesWeight: number;
7267
+ totalWeight: number;
7268
+ items: number;
7269
+ judgedItems: number;
7270
+ topK: {
7271
+ yesWeight: number;
7272
+ totalWeight: number;
7273
+ };
7274
+ }>;
7275
+ /**
7276
+ * The first repetition's items against the judged ones.
7277
+ */
7278
+ changes?: {
7279
+ kept: Array<{
7280
+ key: string;
7281
+ rankBefore?: number;
7282
+ rank?: number;
7283
+ }>;
7284
+ dropped: Array<{
7285
+ key: string;
7286
+ rankBefore?: number;
7287
+ }>;
7288
+ new: Array<{
7289
+ key: string;
7290
+ pointer: string;
7291
+ rank?: number;
7292
+ }>;
7293
+ };
7294
+ /**
7295
+ * The first repetition's tool calls, and what happened to each.
7296
+ */
7297
+ tools?: Array<{
7298
+ step: number;
7299
+ callId: string;
7300
+ toolId: string;
7301
+ toolVersion: string;
7302
+ arguments: unknown;
7303
+ source: "live" | "recorded" | "refused";
7304
+ reason?: string;
7305
+ }>;
7306
+ diverged: boolean;
7307
+ refusedWrites: number;
7308
+ noContext: boolean;
7309
+ approvalSkipped: boolean;
7310
+ error?: string;
7311
+ /**
7312
+ * Set when the replay stopped at a refused write: what it would have done.
7313
+ */
7314
+ stopped?: {
7315
+ toolId: string;
7316
+ arguments: unknown;
7317
+ reason?: string;
7318
+ };
7319
+ };
7320
+ /**
7321
+ * A comparison's `result` (a `judged` eval run's): the summary and each case. `EvalRun.result` stays an open object, since each kind has its own; the clients read it as this (TS `comparisonOf(run)`, Python `comparison_of(run)`).
7322
+ */
7323
+ export type JudgedComparisonResult = {
7324
+ summary: JudgedComparisonSummary;
7325
+ perCase: Array<ComparisonCaseResult>;
7326
+ };
6239
7327
  export type EvalRun = {
6240
7328
  runId: string;
6241
7329
  tenantId: string;
@@ -6249,11 +7337,12 @@ declare namespace Schemas {
6249
7337
  startedAt: string;
6250
7338
  completedAt?: string;
6251
7339
  /**
6252
- * Kind-specific opaque JSON. For `accuracy`, contains `{ passCount, totalCount, meanScore, perCase[] }`. Other kinds define their own shapes as their dispatchers ship.
7340
+ * Kind-specific opaque JSON. For `accuracy`, contains `{ passCount, totalCount, meanScore, perCase[] }`. For `judged` (a comparison), `{ summary, perCase[] }`: the summary has the baseline (the versions behind the recorded runs) and the candidate (`{ kind: "agent", agentId, version }` or `{ kind: "flow", flowId, version, versions? }`), the case counts (`cases`, `diverged`, `refusedWrites`, `errors`, and `stopped`: flow cases that stopped at a write the replay refused, left out of the metrics), the models that answered, and `metrics` (`weightedYesShare`, `judgedCoverage`, `weightedPrecisionAtK`, each `{ baseline, candidate, delta, n, weight, baselineN, baselineWeight, direction, k?, spread? }`); each case has its replay runs, the scores, the items kept, dropped and new, the tool calls with what happened to each, and `stopped` (what it would have done) when it stopped. Other kinds define their own shapes as their dispatchers ship.
6253
7341
  */
6254
7342
  result?: Record<string, unknown>;
6255
7343
  error?: string;
6256
7344
  correlationId?: string;
7345
+ comparison?: EvalComparison;
6257
7346
  };
6258
7347
  export type EvalRunCollectionPage = {
6259
7348
  data: Array<EvalRun>;
@@ -6261,7 +7350,7 @@ declare namespace Schemas {
6261
7350
  hasMore: boolean;
6262
7351
  };
6263
7352
  /**
6264
- * Exactly one of `agentRef` or `flowRef` MUST be supplied. `dryRun: true` returns a plan preview without invoking the subject.
7353
+ * Exactly one of `agentRef` or `flowRef` MUST be supplied. `dryRun: true` returns a plan preview without invoking the subject. For a `judged` suite (a test set), the run is a comparison: `agentRef` or `flowRef` with its `version` is the candidate, replayed on each case without doing anything the past run didn't (a flow stops at a write the replay refuses); `baseline` (default `'recorded'`), `reads` (default `recorded`), `repetitions` (default 1) and `k` (default 10) set how. With `flowRef`, `versions` runs the flow with some of its agents or tools at other versions; an id the flow doesn't use, or a version that isn't published, is refused (`400 validation-failed`, each under `details.issues`).
6265
7354
  */
6266
7355
  export type StartEvalRunBody = {
6267
7356
  /**
@@ -6272,6 +7361,11 @@ declare namespace Schemas {
6272
7361
  flowRef?: EvalRunFlowRef;
6273
7362
  dryRun?: boolean;
6274
7363
  correlationId?: string;
7364
+ baseline?: EvalBaseline;
7365
+ reads?: "recorded" | "live";
7366
+ repetitions?: number;
7367
+ k?: number;
7368
+ versions?: FlowVersionOverrides;
6275
7369
  };
6276
7370
  export type StartEvalRunResult = {
6277
7371
  runId: string;
@@ -6443,16 +7537,48 @@ declare namespace Schemas {
6443
7537
  agents: Array<{
6444
7538
  id: string;
6445
7539
  /**
6446
- * Absent for guardrails, which have no version.
7540
+ * The version the agent is registered as.
6447
7541
  */
6448
- version?: string;
7542
+ version: string;
7543
+ /**
7544
+ * The version the agent's or flow's definition names, present when it differs from `version`: that version was registered already with other pins or content, and versions never change, so the deploy registered the next free version in its line (or an earlier deploy did).
7545
+ */
7546
+ authoredVersion?: string;
7547
+ /**
7548
+ * Why `version` differs from `authoredVersion`: `pins-changed` (a tool or agent it uses has a new version), `unpinned` (`authoredVersion` was published before pins existed), `version-taken` (`authoredVersion` is registered with other content).
7549
+ */
7550
+ reason?: "pins-changed" | "unpinned" | "version-taken";
7551
+ /**
7552
+ * `true`: this deploy registered `version`; `false`: an earlier deploy did.
7553
+ */
7554
+ newVersion?: boolean;
7555
+ /**
7556
+ * For `pins-changed`: the pins that differ from `authoredVersion`'s.
7557
+ */
7558
+ pinChanges?: Array<PinChange>;
6449
7559
  }>;
6450
7560
  flows: Array<{
6451
7561
  id: string;
6452
7562
  /**
6453
- * Absent for guardrails, which have no version.
7563
+ * The version the agent is registered as.
6454
7564
  */
6455
- version?: string;
7565
+ version: string;
7566
+ /**
7567
+ * The version the agent's or flow's definition names, present when it differs from `version`: that version was registered already with other pins or content, and versions never change, so the deploy registered the next free version in its line (or an earlier deploy did).
7568
+ */
7569
+ authoredVersion?: string;
7570
+ /**
7571
+ * Why `version` differs from `authoredVersion`: `pins-changed` (a tool or agent it uses has a new version), `unpinned` (`authoredVersion` was published before pins existed), `version-taken` (`authoredVersion` is registered with other content).
7572
+ */
7573
+ reason?: "pins-changed" | "unpinned" | "version-taken";
7574
+ /**
7575
+ * `true`: this deploy registered `version`; `false`: an earlier deploy did.
7576
+ */
7577
+ newVersion?: boolean;
7578
+ /**
7579
+ * For `pins-changed`: the pins that differ from `authoredVersion`'s.
7580
+ */
7581
+ pinChanges?: Array<PinChange>;
6456
7582
  }>;
6457
7583
  };
6458
7584
  export type DeploymentRecord = {
@@ -7362,13 +8488,160 @@ declare const MintPublicRunTokenResult: z.ZodObject<{
7362
8488
  expiresAt: z.ZodISODateTime;
7363
8489
  runIds: z.ZodArray<z.ZodString>;
7364
8490
  }, z.core.$strict>;
8491
+ export type JudgedEvalCase = Schemas.JudgedEvalCase;
8492
+ declare const JudgedEvalCase: z.ZodObject<{
8493
+ caseId: z.ZodString;
8494
+ subject: z.ZodObject<{
8495
+ kind: z.ZodEnum<{
8496
+ agent: "agent";
8497
+ flow: "flow";
8498
+ }>;
8499
+ id: z.ZodString;
8500
+ version: z.ZodString;
8501
+ }, z.core.$strict>;
8502
+ input: z.ZodUnknown;
8503
+ context: z.ZodOptional<z.ZodObject<{
8504
+ history: z.ZodOptional<z.ZodArray<z.ZodUnknown>>;
8505
+ historyTruncated: z.ZodOptional<z.ZodBoolean>;
8506
+ retrieved: z.ZodOptional<z.ZodUnknown>;
8507
+ sessionApproval: z.ZodOptional<z.ZodObject<{
8508
+ approved: z.ZodBoolean;
8509
+ rationale: z.ZodOptional<z.ZodString>;
8510
+ }, z.core.$strict>>;
8511
+ flow: z.ZodOptional<z.ZodObject<{
8512
+ calls: z.ZodArray<z.ZodObject<{
8513
+ runId: z.ZodString;
8514
+ nodeId: z.ZodOptional<z.ZodString>;
8515
+ scope: z.ZodOptional<z.ZodString>;
8516
+ toolId: z.ZodString;
8517
+ arguments: z.ZodOptional<z.ZodUnknown>;
8518
+ result: z.ZodOptional<z.ZodUnknown>;
8519
+ }, z.core.$strict>>;
8520
+ steps: z.ZodArray<z.ZodObject<{
8521
+ runId: z.ZodString;
8522
+ nodeId: z.ZodOptional<z.ZodString>;
8523
+ scope: z.ZodOptional<z.ZodString>;
8524
+ agentId: z.ZodString;
8525
+ agentVersion: z.ZodString;
8526
+ retrieved: z.ZodOptional<z.ZodUnknown>;
8527
+ }, z.core.$strict>>;
8528
+ truncated: z.ZodOptional<z.ZodBoolean>;
8529
+ }, z.core.$strict>>;
8530
+ }, z.core.$strict>>;
8531
+ output: z.ZodUnknown;
8532
+ items: z.ZodArray<z.ZodObject<{
8533
+ key: z.ZodString;
8534
+ pointer: z.ZodOptional<z.ZodString>;
8535
+ rank: z.ZodOptional<z.ZodNumber>;
8536
+ yes: z.ZodNumber;
8537
+ no: z.ZodNumber;
8538
+ yesWeight: z.ZodNumber;
8539
+ totalWeight: z.ZodNumber;
8540
+ reasons: z.ZodArray<z.ZodObject<{
8541
+ verdict: z.ZodEnum<{
8542
+ yes: "yes";
8543
+ no: "no";
8544
+ }>;
8545
+ reason: z.ZodString;
8546
+ }, z.core.$strict>>;
8547
+ }, z.core.$strict>>;
8548
+ }, z.core.$strict>;
8549
+ export type JudgedEvalCaseCollectionPage = Schemas.JudgedEvalCaseCollectionPage;
8550
+ declare const JudgedEvalCaseCollectionPage: z.ZodObject<{
8551
+ data: z.ZodArray<z.ZodObject<{
8552
+ caseId: z.ZodString;
8553
+ subject: z.ZodObject<{
8554
+ kind: z.ZodEnum<{
8555
+ agent: "agent";
8556
+ flow: "flow";
8557
+ }>;
8558
+ id: z.ZodString;
8559
+ version: z.ZodString;
8560
+ }, z.core.$strict>;
8561
+ input: z.ZodUnknown;
8562
+ context: z.ZodOptional<z.ZodObject<{
8563
+ history: z.ZodOptional<z.ZodArray<z.ZodUnknown>>;
8564
+ historyTruncated: z.ZodOptional<z.ZodBoolean>;
8565
+ retrieved: z.ZodOptional<z.ZodUnknown>;
8566
+ sessionApproval: z.ZodOptional<z.ZodObject<{
8567
+ approved: z.ZodBoolean;
8568
+ rationale: z.ZodOptional<z.ZodString>;
8569
+ }, z.core.$strict>>;
8570
+ flow: z.ZodOptional<z.ZodObject<{
8571
+ calls: z.ZodArray<z.ZodObject<{
8572
+ runId: z.ZodString;
8573
+ nodeId: z.ZodOptional<z.ZodString>;
8574
+ scope: z.ZodOptional<z.ZodString>;
8575
+ toolId: z.ZodString;
8576
+ arguments: z.ZodOptional<z.ZodUnknown>;
8577
+ result: z.ZodOptional<z.ZodUnknown>;
8578
+ }, z.core.$strict>>;
8579
+ steps: z.ZodArray<z.ZodObject<{
8580
+ runId: z.ZodString;
8581
+ nodeId: z.ZodOptional<z.ZodString>;
8582
+ scope: z.ZodOptional<z.ZodString>;
8583
+ agentId: z.ZodString;
8584
+ agentVersion: z.ZodString;
8585
+ retrieved: z.ZodOptional<z.ZodUnknown>;
8586
+ }, z.core.$strict>>;
8587
+ truncated: z.ZodOptional<z.ZodBoolean>;
8588
+ }, z.core.$strict>>;
8589
+ }, z.core.$strict>>;
8590
+ output: z.ZodUnknown;
8591
+ items: z.ZodArray<z.ZodObject<{
8592
+ key: z.ZodString;
8593
+ pointer: z.ZodOptional<z.ZodString>;
8594
+ rank: z.ZodOptional<z.ZodNumber>;
8595
+ yes: z.ZodNumber;
8596
+ no: z.ZodNumber;
8597
+ yesWeight: z.ZodNumber;
8598
+ totalWeight: z.ZodNumber;
8599
+ reasons: z.ZodArray<z.ZodObject<{
8600
+ verdict: z.ZodEnum<{
8601
+ yes: "yes";
8602
+ no: "no";
8603
+ }>;
8604
+ reason: z.ZodString;
8605
+ }, z.core.$strict>>;
8606
+ }, z.core.$strict>>;
8607
+ }, z.core.$strict>>;
8608
+ nextCursor: z.ZodOptional<z.ZodString>;
8609
+ hasMore: z.ZodBoolean;
8610
+ }, z.core.$strict>;
8611
+ export type BuildJudgedSuiteBody = Schemas.BuildJudgedSuiteBody;
8612
+ declare const BuildJudgedSuiteBody: z.ZodObject<{
8613
+ version: z.ZodString;
8614
+ projectId: z.ZodString;
8615
+ agentId: z.ZodOptional<z.ZodString>;
8616
+ agentVersion: z.ZodOptional<z.ZodString>;
8617
+ flowId: z.ZodOptional<z.ZodString>;
8618
+ since: z.ZodOptional<z.ZodISODateTime>;
8619
+ until: z.ZodOptional<z.ZodISODateTime>;
8620
+ judgeClassIds: z.ZodOptional<z.ZodArray<z.ZodString>>;
8621
+ minJudgments: z.ZodOptional<z.ZodNumber>;
8622
+ description: z.ZodOptional<z.ZodString>;
8623
+ }, z.core.$strict>;
8624
+ export type BuildJudgedSuiteResult = Schemas.BuildJudgedSuiteResult;
8625
+ declare const BuildJudgedSuiteResult: z.ZodObject<{
8626
+ suiteId: z.ZodString;
8627
+ version: z.ZodString;
8628
+ kind: z.ZodLiteral<"judged">;
8629
+ caseCount: z.ZodNumber;
8630
+ truncated: z.ZodBoolean;
8631
+ }, z.core.$strict>;
7365
8632
  type Agent$1 = Schemas.Agent;
7366
8633
  declare const Agent$1: z.ZodObject<{
7367
8634
  id: z.ZodString;
7368
8635
  version: z.ZodString;
7369
8636
  name: z.ZodString;
7370
8637
  description: z.ZodOptional<z.ZodString>;
7371
- instructions: z.ZodString;
8638
+ instructions: z.ZodUnion<readonly [
8639
+ z.ZodString,
8640
+ z.ZodObject<{
8641
+ prompt: z.ZodString;
8642
+ version: z.ZodString;
8643
+ }, z.core.$strict>
8644
+ ]>;
7372
8645
  parameters: z.ZodOptional<z.ZodArray<z.ZodObject<{
7373
8646
  name: z.ZodString;
7374
8647
  description: z.ZodOptional<z.ZodString>;
@@ -7381,6 +8654,14 @@ declare const Agent$1: z.ZodObject<{
7381
8654
  required: z.ZodOptional<z.ZodBoolean>;
7382
8655
  default: z.ZodOptional<z.ZodUnknown>;
7383
8656
  }, z.core.$strict>>>;
8657
+ settings: z.ZodOptional<z.ZodArray<z.ZodObject<{
8658
+ id: z.ZodString;
8659
+ version: z.ZodString;
8660
+ }, z.core.$strict>>>;
8661
+ modelSettings: z.ZodOptional<z.ZodObject<{
8662
+ id: z.ZodString;
8663
+ version: z.ZodString;
8664
+ }, z.core.$strict>>;
7384
8665
  capabilities: z.ZodArray<z.ZodRecord<z.ZodString, z.ZodUnknown>>;
7385
8666
  tools: z.ZodArray<z.ZodObject<{
7386
8667
  id: z.ZodString;
@@ -7427,6 +8708,34 @@ declare const Agent$1: z.ZodObject<{
7427
8708
  "unknown-tool": "unknown-tool";
7428
8709
  }>>>;
7429
8710
  }, z.core.$strict>>;
8711
+ pins: z.ZodOptional<z.ZodObject<{
8712
+ tools: z.ZodRecord<z.ZodString, z.ZodString>;
8713
+ prompts: z.ZodRecord<z.ZodString, z.ZodString>;
8714
+ settings: z.ZodRecord<z.ZodString, z.ZodString>;
8715
+ }, z.core.$strict>>;
8716
+ derivedFrom: z.ZodOptional<z.ZodObject<{
8717
+ version: z.ZodString;
8718
+ reason: z.ZodEnum<{
8719
+ "pins-changed": "pins-changed";
8720
+ unpinned: "unpinned";
8721
+ "version-taken": "version-taken";
8722
+ edited: "edited";
8723
+ }>;
8724
+ label: z.ZodOptional<z.ZodString>;
8725
+ by: z.ZodOptional<z.ZodString>;
8726
+ }, z.core.$strict>>;
8727
+ unregisteredAt: z.ZodOptional<z.ZodISODateTime>;
8728
+ pinsDigest: z.ZodOptional<z.ZodString>;
8729
+ }, z.core.$strict>;
8730
+ export type DeriveAgentVersionBody = Schemas.DeriveAgentVersionBody;
8731
+ declare const DeriveAgentVersionBody: z.ZodObject<{
8732
+ from: z.ZodString;
8733
+ pins: z.ZodObject<{
8734
+ prompts: z.ZodOptional<z.ZodRecord<z.ZodString, z.ZodString>>;
8735
+ settings: z.ZodOptional<z.ZodRecord<z.ZodString, z.ZodString>>;
8736
+ }, z.core.$strict>;
8737
+ label: z.ZodOptional<z.ZodString>;
8738
+ projectId: z.ZodOptional<z.ZodUUID>;
7430
8739
  }, z.core.$strict>;
7431
8740
  export type UnregisterAgentResult = Schemas.UnregisterAgentResult;
7432
8741
  declare const UnregisterAgentResult: z.ZodObject<{
@@ -7441,7 +8750,13 @@ declare const AgentCollectionPage: z.ZodObject<{
7441
8750
  version: z.ZodString;
7442
8751
  name: z.ZodString;
7443
8752
  description: z.ZodOptional<z.ZodString>;
7444
- instructions: z.ZodString;
8753
+ instructions: z.ZodUnion<readonly [
8754
+ z.ZodString,
8755
+ z.ZodObject<{
8756
+ prompt: z.ZodString;
8757
+ version: z.ZodString;
8758
+ }, z.core.$strict>
8759
+ ]>;
7445
8760
  parameters: z.ZodOptional<z.ZodArray<z.ZodObject<{
7446
8761
  name: z.ZodString;
7447
8762
  description: z.ZodOptional<z.ZodString>;
@@ -7454,6 +8769,14 @@ declare const AgentCollectionPage: z.ZodObject<{
7454
8769
  required: z.ZodOptional<z.ZodBoolean>;
7455
8770
  default: z.ZodOptional<z.ZodUnknown>;
7456
8771
  }, z.core.$strict>>>;
8772
+ settings: z.ZodOptional<z.ZodArray<z.ZodObject<{
8773
+ id: z.ZodString;
8774
+ version: z.ZodString;
8775
+ }, z.core.$strict>>>;
8776
+ modelSettings: z.ZodOptional<z.ZodObject<{
8777
+ id: z.ZodString;
8778
+ version: z.ZodString;
8779
+ }, z.core.$strict>>;
7457
8780
  capabilities: z.ZodArray<z.ZodRecord<z.ZodString, z.ZodUnknown>>;
7458
8781
  tools: z.ZodArray<z.ZodObject<{
7459
8782
  id: z.ZodString;
@@ -7500,6 +8823,24 @@ declare const AgentCollectionPage: z.ZodObject<{
7500
8823
  "unknown-tool": "unknown-tool";
7501
8824
  }>>>;
7502
8825
  }, z.core.$strict>>;
8826
+ pins: z.ZodOptional<z.ZodObject<{
8827
+ tools: z.ZodRecord<z.ZodString, z.ZodString>;
8828
+ prompts: z.ZodRecord<z.ZodString, z.ZodString>;
8829
+ settings: z.ZodRecord<z.ZodString, z.ZodString>;
8830
+ }, z.core.$strict>>;
8831
+ derivedFrom: z.ZodOptional<z.ZodObject<{
8832
+ version: z.ZodString;
8833
+ reason: z.ZodEnum<{
8834
+ "pins-changed": "pins-changed";
8835
+ unpinned: "unpinned";
8836
+ "version-taken": "version-taken";
8837
+ edited: "edited";
8838
+ }>;
8839
+ label: z.ZodOptional<z.ZodString>;
8840
+ by: z.ZodOptional<z.ZodString>;
8841
+ }, z.core.$strict>>;
8842
+ unregisteredAt: z.ZodOptional<z.ZodISODateTime>;
8843
+ pinsDigest: z.ZodOptional<z.ZodString>;
7503
8844
  }, z.core.$strict>>;
7504
8845
  nextCursor: z.ZodOptional<z.ZodString>;
7505
8846
  hasMore: z.ZodBoolean;
@@ -7538,6 +8879,7 @@ declare const ProviderMetadata: z.ZodObject<{
7538
8879
  description: z.ZodOptional<z.ZodString>;
7539
8880
  capabilityKind: z.ZodOptional<z.ZodString>;
7540
8881
  fallback: z.ZodOptional<z.ZodBoolean>;
8882
+ labels: z.ZodOptional<z.ZodRecord<z.ZodString, z.ZodString>>;
7541
8883
  }, z.core.$strict>;
7542
8884
  export type ProviderCollectionPage = Schemas.ProviderCollectionPage;
7543
8885
  declare const ProviderCollectionPage: z.ZodObject<{
@@ -7574,6 +8916,7 @@ declare const ProviderCollectionPage: z.ZodObject<{
7574
8916
  description: z.ZodOptional<z.ZodString>;
7575
8917
  capabilityKind: z.ZodOptional<z.ZodString>;
7576
8918
  fallback: z.ZodOptional<z.ZodBoolean>;
8919
+ labels: z.ZodOptional<z.ZodRecord<z.ZodString, z.ZodString>>;
7577
8920
  }, z.core.$strict>>;
7578
8921
  nextCursor: z.ZodOptional<z.ZodString>;
7579
8922
  hasMore: z.ZodBoolean;
@@ -7613,6 +8956,7 @@ declare const RegisterProviderBody: z.ZodObject<{
7613
8956
  description: z.ZodOptional<z.ZodString>;
7614
8957
  capabilityKind: z.ZodOptional<z.ZodString>;
7615
8958
  fallback: z.ZodOptional<z.ZodBoolean>;
8959
+ labels: z.ZodOptional<z.ZodRecord<z.ZodString, z.ZodString>>;
7616
8960
  }, z.core.$strict>;
7617
8961
  adapter_id: z.ZodString;
7618
8962
  secret_ref: z.ZodOptional<z.ZodObject<{
@@ -7675,6 +9019,7 @@ declare const EvalKind: z.ZodEnum<{
7675
9019
  "human-review": "human-review";
7676
9020
  benchmark: "benchmark";
7677
9021
  custom: "custom";
9022
+ judged: "judged";
7678
9023
  }>;
7679
9024
  export type EvalSuite = Schemas.EvalSuite;
7680
9025
  declare const EvalSuite: z.ZodObject<{
@@ -7688,6 +9033,7 @@ declare const EvalSuite: z.ZodObject<{
7688
9033
  "human-review": "human-review";
7689
9034
  benchmark: "benchmark";
7690
9035
  custom: "custom";
9036
+ judged: "judged";
7691
9037
  }>;
7692
9038
  description: z.ZodOptional<z.ZodString>;
7693
9039
  spec: z.ZodRecord<z.ZodString, z.ZodUnknown>;
@@ -7705,6 +9051,7 @@ declare const EvalSuiteCollectionPage: z.ZodObject<{
7705
9051
  "human-review": "human-review";
7706
9052
  benchmark: "benchmark";
7707
9053
  custom: "custom";
9054
+ judged: "judged";
7708
9055
  }>;
7709
9056
  description: z.ZodOptional<z.ZodString>;
7710
9057
  spec: z.ZodRecord<z.ZodString, z.ZodUnknown>;
@@ -7712,37 +9059,164 @@ declare const EvalSuiteCollectionPage: z.ZodObject<{
7712
9059
  nextCursor: z.ZodOptional<z.ZodString>;
7713
9060
  hasMore: z.ZodBoolean;
7714
9061
  }, z.core.$strict>;
7715
- export type PublishEvalSuiteBody = Schemas.PublishEvalSuiteBody;
7716
- declare const PublishEvalSuiteBody: z.ZodObject<{
7717
- id: z.ZodString;
9062
+ export type PublishEvalSuiteBody = Schemas.PublishEvalSuiteBody;
9063
+ declare const PublishEvalSuiteBody: z.ZodObject<{
9064
+ id: z.ZodString;
9065
+ projectId: z.ZodUUID;
9066
+ tenantId: z.ZodOptional<z.ZodUUID>;
9067
+ version: z.ZodString;
9068
+ kind: z.ZodEnum<{
9069
+ accuracy: "accuracy";
9070
+ pairwise: "pairwise";
9071
+ regression: "regression";
9072
+ "human-review": "human-review";
9073
+ benchmark: "benchmark";
9074
+ custom: "custom";
9075
+ judged: "judged";
9076
+ }>;
9077
+ description: z.ZodOptional<z.ZodString>;
9078
+ spec: z.ZodRecord<z.ZodString, z.ZodUnknown>;
9079
+ }, z.core.$strict>;
9080
+ export type PublishEvalSuiteResult = Schemas.PublishEvalSuiteResult;
9081
+ declare const PublishEvalSuiteResult: z.ZodObject<{
9082
+ suiteId: z.ZodString;
9083
+ version: z.ZodString;
9084
+ }, z.core.$strict>;
9085
+ export type UnregisterEvalSuiteResult = Schemas.UnregisterEvalSuiteResult;
9086
+ declare const UnregisterEvalSuiteResult: z.ZodObject<{
9087
+ suiteId: z.ZodString;
9088
+ version: z.ZodString;
9089
+ unregistered: z.ZodLiteral<true>;
9090
+ }, z.core.$strict>;
9091
+ export type ReinstateEvalSuiteVersionResult = Schemas.ReinstateEvalSuiteVersionResult;
9092
+ declare const ReinstateEvalSuiteVersionResult: z.ZodObject<{
9093
+ suiteId: z.ZodString;
9094
+ version: z.ZodString;
9095
+ wasTombstoned: z.ZodBoolean;
9096
+ }, z.core.$strict>;
9097
+ export type BlockKind = Schemas.BlockKind;
9098
+ export declare const BlockKind: z.ZodEnum<{
9099
+ prompt: "prompt";
9100
+ settings: "settings";
9101
+ }>;
9102
+ export type Block = Schemas.Block;
9103
+ export declare const Block: z.ZodObject<{
9104
+ id: z.ZodString;
9105
+ version: z.ZodString;
9106
+ kind: z.ZodEnum<{
9107
+ prompt: "prompt";
9108
+ settings: "settings";
9109
+ }>;
9110
+ description: z.ZodOptional<z.ZodString>;
9111
+ content: z.ZodUnion<readonly [
9112
+ z.ZodObject<{
9113
+ template: z.ZodString;
9114
+ parameters: z.ZodOptional<z.ZodArray<z.ZodObject<{
9115
+ name: z.ZodString;
9116
+ description: z.ZodOptional<z.ZodString>;
9117
+ type: z.ZodEnum<{
9118
+ string: "string";
9119
+ number: "number";
9120
+ boolean: "boolean";
9121
+ date: "date";
9122
+ }>;
9123
+ required: z.ZodOptional<z.ZodBoolean>;
9124
+ default: z.ZodOptional<z.ZodUnknown>;
9125
+ }, z.core.$strict>>>;
9126
+ }, z.core.$strict>,
9127
+ z.ZodObject<{
9128
+ values: z.ZodRecord<z.ZodString, z.ZodUnknown>;
9129
+ schema: z.ZodOptional<z.ZodRecord<z.ZodString, z.ZodUnknown>>;
9130
+ }, z.core.$strict>
9131
+ ]>;
9132
+ projectId: z.ZodUUID;
9133
+ publishedAt: z.ZodISODateTime;
9134
+ unregisteredAt: z.ZodOptional<z.ZodISODateTime>;
9135
+ }, z.core.$strict>;
9136
+ export type BlockCollectionPage = Schemas.BlockCollectionPage;
9137
+ declare const BlockCollectionPage: z.ZodObject<{
9138
+ data: z.ZodArray<z.ZodObject<{
9139
+ id: z.ZodString;
9140
+ version: z.ZodString;
9141
+ kind: z.ZodEnum<{
9142
+ prompt: "prompt";
9143
+ settings: "settings";
9144
+ }>;
9145
+ description: z.ZodOptional<z.ZodString>;
9146
+ content: z.ZodUnion<readonly [
9147
+ z.ZodObject<{
9148
+ template: z.ZodString;
9149
+ parameters: z.ZodOptional<z.ZodArray<z.ZodObject<{
9150
+ name: z.ZodString;
9151
+ description: z.ZodOptional<z.ZodString>;
9152
+ type: z.ZodEnum<{
9153
+ string: "string";
9154
+ number: "number";
9155
+ boolean: "boolean";
9156
+ date: "date";
9157
+ }>;
9158
+ required: z.ZodOptional<z.ZodBoolean>;
9159
+ default: z.ZodOptional<z.ZodUnknown>;
9160
+ }, z.core.$strict>>>;
9161
+ }, z.core.$strict>,
9162
+ z.ZodObject<{
9163
+ values: z.ZodRecord<z.ZodString, z.ZodUnknown>;
9164
+ schema: z.ZodOptional<z.ZodRecord<z.ZodString, z.ZodUnknown>>;
9165
+ }, z.core.$strict>
9166
+ ]>;
9167
+ projectId: z.ZodUUID;
9168
+ publishedAt: z.ZodISODateTime;
9169
+ unregisteredAt: z.ZodOptional<z.ZodISODateTime>;
9170
+ }, z.core.$strict>>;
9171
+ nextCursor: z.ZodOptional<z.ZodString>;
9172
+ hasMore: z.ZodBoolean;
9173
+ }, z.core.$strict>;
9174
+ export type PublishBlockBody = Schemas.PublishBlockBody;
9175
+ declare const PublishBlockBody: z.ZodObject<{
7718
9176
  projectId: z.ZodUUID;
7719
- tenantId: z.ZodOptional<z.ZodUUID>;
9177
+ id: z.ZodString;
7720
9178
  version: z.ZodString;
7721
9179
  kind: z.ZodEnum<{
7722
- accuracy: "accuracy";
7723
- pairwise: "pairwise";
7724
- regression: "regression";
7725
- "human-review": "human-review";
7726
- benchmark: "benchmark";
7727
- custom: "custom";
9180
+ prompt: "prompt";
9181
+ settings: "settings";
7728
9182
  }>;
7729
9183
  description: z.ZodOptional<z.ZodString>;
7730
- spec: z.ZodRecord<z.ZodString, z.ZodUnknown>;
9184
+ content: z.ZodUnion<readonly [
9185
+ z.ZodObject<{
9186
+ template: z.ZodString;
9187
+ parameters: z.ZodOptional<z.ZodArray<z.ZodObject<{
9188
+ name: z.ZodString;
9189
+ description: z.ZodOptional<z.ZodString>;
9190
+ type: z.ZodEnum<{
9191
+ string: "string";
9192
+ number: "number";
9193
+ boolean: "boolean";
9194
+ date: "date";
9195
+ }>;
9196
+ required: z.ZodOptional<z.ZodBoolean>;
9197
+ default: z.ZodOptional<z.ZodUnknown>;
9198
+ }, z.core.$strict>>>;
9199
+ }, z.core.$strict>,
9200
+ z.ZodObject<{
9201
+ values: z.ZodRecord<z.ZodString, z.ZodUnknown>;
9202
+ schema: z.ZodOptional<z.ZodRecord<z.ZodString, z.ZodUnknown>>;
9203
+ }, z.core.$strict>
9204
+ ]>;
7731
9205
  }, z.core.$strict>;
7732
- export type PublishEvalSuiteResult = Schemas.PublishEvalSuiteResult;
7733
- declare const PublishEvalSuiteResult: z.ZodObject<{
7734
- suiteId: z.ZodString;
9206
+ export type PublishBlockResult = Schemas.PublishBlockResult;
9207
+ export declare const PublishBlockResult: z.ZodObject<{
9208
+ blockId: z.ZodString;
7735
9209
  version: z.ZodString;
7736
9210
  }, z.core.$strict>;
7737
- export type UnregisterEvalSuiteResult = Schemas.UnregisterEvalSuiteResult;
7738
- declare const UnregisterEvalSuiteResult: z.ZodObject<{
7739
- suiteId: z.ZodString;
9211
+ export type UnregisterBlockResult = Schemas.UnregisterBlockResult;
9212
+ export declare const UnregisterBlockResult: z.ZodObject<{
9213
+ blockId: z.ZodString;
7740
9214
  version: z.ZodString;
7741
9215
  unregistered: z.ZodLiteral<true>;
7742
9216
  }, z.core.$strict>;
7743
- export type ReinstateEvalSuiteVersionResult = Schemas.ReinstateEvalSuiteVersionResult;
7744
- declare const ReinstateEvalSuiteVersionResult: z.ZodObject<{
7745
- suiteId: z.ZodString;
9217
+ export type ReinstateBlockResult = Schemas.ReinstateBlockResult;
9218
+ export declare const ReinstateBlockResult: z.ZodObject<{
9219
+ blockId: z.ZodString;
7746
9220
  version: z.ZodString;
7747
9221
  wasTombstoned: z.ZodBoolean;
7748
9222
  }, z.core.$strict>;
@@ -7754,6 +9228,383 @@ declare const EvalRunStatus: z.ZodEnum<{
7754
9228
  completed: "completed";
7755
9229
  cancelled: "cancelled";
7756
9230
  }>;
9231
+ export type ComparisonMetric = Schemas.ComparisonMetric;
9232
+ export declare const ComparisonMetric: z.ZodObject<{
9233
+ baseline: z.ZodNullable<z.ZodNumber>;
9234
+ candidate: z.ZodNullable<z.ZodNumber>;
9235
+ delta: z.ZodNullable<z.ZodNumber>;
9236
+ n: z.ZodNumber;
9237
+ weight: z.ZodNumber;
9238
+ baselineN: z.ZodNumber;
9239
+ baselineWeight: z.ZodNumber;
9240
+ direction: z.ZodLiteral<"higher">;
9241
+ k: z.ZodOptional<z.ZodNumber>;
9242
+ spread: z.ZodOptional<z.ZodNumber>;
9243
+ }, z.core.$strict>;
9244
+ export type ComparisonCandidate = Schemas.ComparisonCandidate;
9245
+ export declare const ComparisonCandidate: z.ZodUnion<readonly [
9246
+ z.ZodObject<{
9247
+ kind: z.ZodLiteral<"agent">;
9248
+ agentId: z.ZodString;
9249
+ version: z.ZodString;
9250
+ }, z.core.$strict>,
9251
+ z.ZodObject<{
9252
+ kind: z.ZodLiteral<"flow">;
9253
+ flowId: z.ZodString;
9254
+ version: z.ZodString;
9255
+ versions: z.ZodOptional<z.ZodObject<{
9256
+ tools: z.ZodOptional<z.ZodRecord<z.ZodString, z.ZodString>>;
9257
+ agents: z.ZodOptional<z.ZodRecord<z.ZodString, z.ZodString>>;
9258
+ }, z.core.$strict>>;
9259
+ }, z.core.$strict>
9260
+ ]>;
9261
+ export type JudgedComparisonSummary = Schemas.JudgedComparisonSummary;
9262
+ export declare const JudgedComparisonSummary: z.ZodObject<{
9263
+ evalRunId: z.ZodString;
9264
+ status: z.ZodEnum<{
9265
+ failed: "failed";
9266
+ completed: "completed";
9267
+ partial: "partial";
9268
+ }>;
9269
+ completedAt: z.ZodISODateTime;
9270
+ suite: z.ZodObject<{
9271
+ id: z.ZodString;
9272
+ version: z.ZodString;
9273
+ }, z.core.$strict>;
9274
+ candidate: z.ZodUnion<readonly [
9275
+ z.ZodObject<{
9276
+ kind: z.ZodLiteral<"agent">;
9277
+ agentId: z.ZodString;
9278
+ version: z.ZodString;
9279
+ }, z.core.$strict>,
9280
+ z.ZodObject<{
9281
+ kind: z.ZodLiteral<"flow">;
9282
+ flowId: z.ZodString;
9283
+ version: z.ZodString;
9284
+ versions: z.ZodOptional<z.ZodObject<{
9285
+ tools: z.ZodOptional<z.ZodRecord<z.ZodString, z.ZodString>>;
9286
+ agents: z.ZodOptional<z.ZodRecord<z.ZodString, z.ZodString>>;
9287
+ }, z.core.$strict>>;
9288
+ }, z.core.$strict>
9289
+ ]>;
9290
+ baseline: z.ZodUnion<readonly [
9291
+ z.ZodObject<{
9292
+ kind: z.ZodLiteral<"recorded">;
9293
+ versions: z.ZodArray<z.ZodObject<{
9294
+ agentId: z.ZodOptional<z.ZodString>;
9295
+ flowId: z.ZodOptional<z.ZodString>;
9296
+ version: z.ZodString;
9297
+ cases: z.ZodNumber;
9298
+ }, z.core.$strict>>;
9299
+ }, z.core.$strict>,
9300
+ z.ZodObject<{
9301
+ kind: z.ZodLiteral<"version">;
9302
+ agentId: z.ZodString;
9303
+ version: z.ZodString;
9304
+ via: z.ZodEnum<{
9305
+ live: "live";
9306
+ explicit: "explicit";
9307
+ }>;
9308
+ liveScope: z.ZodOptional<z.ZodRecord<z.ZodString, z.ZodUnknown>>;
9309
+ }, z.core.$strict>
9310
+ ]>;
9311
+ scope: z.ZodObject<{
9312
+ projectId: z.ZodOptional<z.ZodString>;
9313
+ }, z.core.$strict>;
9314
+ cases: z.ZodNumber;
9315
+ diverged: z.ZodNumber;
9316
+ refusedWrites: z.ZodNumber;
9317
+ errors: z.ZodNumber;
9318
+ stopped: z.ZodNumber;
9319
+ reads: z.ZodEnum<{
9320
+ recorded: "recorded";
9321
+ live: "live";
9322
+ }>;
9323
+ sampling: z.ZodObject<{
9324
+ models: z.ZodArray<z.ZodObject<{
9325
+ providerId: z.ZodString;
9326
+ model: z.ZodString;
9327
+ runs: z.ZodNumber;
9328
+ }, z.core.$strict>>;
9329
+ }, z.core.$strict>;
9330
+ repetitions: z.ZodNumber;
9331
+ metrics: z.ZodObject<{
9332
+ weightedYesShare: z.ZodObject<{
9333
+ baseline: z.ZodNullable<z.ZodNumber>;
9334
+ candidate: z.ZodNullable<z.ZodNumber>;
9335
+ delta: z.ZodNullable<z.ZodNumber>;
9336
+ n: z.ZodNumber;
9337
+ weight: z.ZodNumber;
9338
+ baselineN: z.ZodNumber;
9339
+ baselineWeight: z.ZodNumber;
9340
+ direction: z.ZodLiteral<"higher">;
9341
+ k: z.ZodOptional<z.ZodNumber>;
9342
+ spread: z.ZodOptional<z.ZodNumber>;
9343
+ }, z.core.$strict>;
9344
+ judgedCoverage: z.ZodObject<{
9345
+ baseline: z.ZodNullable<z.ZodNumber>;
9346
+ candidate: z.ZodNullable<z.ZodNumber>;
9347
+ delta: z.ZodNullable<z.ZodNumber>;
9348
+ n: z.ZodNumber;
9349
+ weight: z.ZodNumber;
9350
+ baselineN: z.ZodNumber;
9351
+ baselineWeight: z.ZodNumber;
9352
+ direction: z.ZodLiteral<"higher">;
9353
+ k: z.ZodOptional<z.ZodNumber>;
9354
+ spread: z.ZodOptional<z.ZodNumber>;
9355
+ }, z.core.$strict>;
9356
+ weightedPrecisionAtK: z.ZodObject<{
9357
+ baseline: z.ZodNullable<z.ZodNumber>;
9358
+ candidate: z.ZodNullable<z.ZodNumber>;
9359
+ delta: z.ZodNullable<z.ZodNumber>;
9360
+ n: z.ZodNumber;
9361
+ weight: z.ZodNumber;
9362
+ baselineN: z.ZodNumber;
9363
+ baselineWeight: z.ZodNumber;
9364
+ direction: z.ZodLiteral<"higher">;
9365
+ k: z.ZodOptional<z.ZodNumber>;
9366
+ spread: z.ZodOptional<z.ZodNumber>;
9367
+ }, z.core.$strict>;
9368
+ }, z.core.$strict>;
9369
+ }, z.core.$strict>;
9370
+ export type ComparisonCaseResult = Schemas.ComparisonCaseResult;
9371
+ export declare const ComparisonCaseResult: z.ZodObject<{
9372
+ caseId: z.ZodString;
9373
+ runIds: z.ZodArray<z.ZodString>;
9374
+ baseline: z.ZodObject<{
9375
+ yesWeight: z.ZodNumber;
9376
+ totalWeight: z.ZodNumber;
9377
+ items: z.ZodNumber;
9378
+ judgedItems: z.ZodNumber;
9379
+ topK: z.ZodObject<{
9380
+ yesWeight: z.ZodNumber;
9381
+ totalWeight: z.ZodNumber;
9382
+ }, z.core.$strict>;
9383
+ }, z.core.$strict>;
9384
+ candidate: z.ZodArray<z.ZodObject<{
9385
+ yesWeight: z.ZodNumber;
9386
+ totalWeight: z.ZodNumber;
9387
+ items: z.ZodNumber;
9388
+ judgedItems: z.ZodNumber;
9389
+ topK: z.ZodObject<{
9390
+ yesWeight: z.ZodNumber;
9391
+ totalWeight: z.ZodNumber;
9392
+ }, z.core.$strict>;
9393
+ }, z.core.$strict>>;
9394
+ changes: z.ZodOptional<z.ZodObject<{
9395
+ kept: z.ZodArray<z.ZodObject<{
9396
+ key: z.ZodString;
9397
+ rankBefore: z.ZodOptional<z.ZodNumber>;
9398
+ rank: z.ZodOptional<z.ZodNumber>;
9399
+ }, z.core.$strict>>;
9400
+ dropped: z.ZodArray<z.ZodObject<{
9401
+ key: z.ZodString;
9402
+ rankBefore: z.ZodOptional<z.ZodNumber>;
9403
+ }, z.core.$strict>>;
9404
+ new: z.ZodArray<z.ZodObject<{
9405
+ key: z.ZodString;
9406
+ pointer: z.ZodString;
9407
+ rank: z.ZodOptional<z.ZodNumber>;
9408
+ }, z.core.$strict>>;
9409
+ }, z.core.$strict>>;
9410
+ tools: z.ZodOptional<z.ZodArray<z.ZodObject<{
9411
+ step: z.ZodNumber;
9412
+ callId: z.ZodString;
9413
+ toolId: z.ZodString;
9414
+ toolVersion: z.ZodString;
9415
+ arguments: z.ZodUnknown;
9416
+ source: z.ZodEnum<{
9417
+ recorded: "recorded";
9418
+ live: "live";
9419
+ refused: "refused";
9420
+ }>;
9421
+ reason: z.ZodOptional<z.ZodString>;
9422
+ }, z.core.$strict>>>;
9423
+ diverged: z.ZodBoolean;
9424
+ refusedWrites: z.ZodNumber;
9425
+ noContext: z.ZodBoolean;
9426
+ approvalSkipped: z.ZodBoolean;
9427
+ error: z.ZodOptional<z.ZodString>;
9428
+ stopped: z.ZodOptional<z.ZodObject<{
9429
+ toolId: z.ZodString;
9430
+ arguments: z.ZodUnknown;
9431
+ reason: z.ZodOptional<z.ZodString>;
9432
+ }, z.core.$strict>>;
9433
+ }, z.core.$strict>;
9434
+ export type JudgedComparisonResult = Schemas.JudgedComparisonResult;
9435
+ export declare const JudgedComparisonResult: z.ZodObject<{
9436
+ summary: z.ZodObject<{
9437
+ evalRunId: z.ZodString;
9438
+ status: z.ZodEnum<{
9439
+ failed: "failed";
9440
+ completed: "completed";
9441
+ partial: "partial";
9442
+ }>;
9443
+ completedAt: z.ZodISODateTime;
9444
+ suite: z.ZodObject<{
9445
+ id: z.ZodString;
9446
+ version: z.ZodString;
9447
+ }, z.core.$strict>;
9448
+ candidate: z.ZodUnion<readonly [
9449
+ z.ZodObject<{
9450
+ kind: z.ZodLiteral<"agent">;
9451
+ agentId: z.ZodString;
9452
+ version: z.ZodString;
9453
+ }, z.core.$strict>,
9454
+ z.ZodObject<{
9455
+ kind: z.ZodLiteral<"flow">;
9456
+ flowId: z.ZodString;
9457
+ version: z.ZodString;
9458
+ versions: z.ZodOptional<z.ZodObject<{
9459
+ tools: z.ZodOptional<z.ZodRecord<z.ZodString, z.ZodString>>;
9460
+ agents: z.ZodOptional<z.ZodRecord<z.ZodString, z.ZodString>>;
9461
+ }, z.core.$strict>>;
9462
+ }, z.core.$strict>
9463
+ ]>;
9464
+ baseline: z.ZodUnion<readonly [
9465
+ z.ZodObject<{
9466
+ kind: z.ZodLiteral<"recorded">;
9467
+ versions: z.ZodArray<z.ZodObject<{
9468
+ agentId: z.ZodOptional<z.ZodString>;
9469
+ flowId: z.ZodOptional<z.ZodString>;
9470
+ version: z.ZodString;
9471
+ cases: z.ZodNumber;
9472
+ }, z.core.$strict>>;
9473
+ }, z.core.$strict>,
9474
+ z.ZodObject<{
9475
+ kind: z.ZodLiteral<"version">;
9476
+ agentId: z.ZodString;
9477
+ version: z.ZodString;
9478
+ via: z.ZodEnum<{
9479
+ live: "live";
9480
+ explicit: "explicit";
9481
+ }>;
9482
+ liveScope: z.ZodOptional<z.ZodRecord<z.ZodString, z.ZodUnknown>>;
9483
+ }, z.core.$strict>
9484
+ ]>;
9485
+ scope: z.ZodObject<{
9486
+ projectId: z.ZodOptional<z.ZodString>;
9487
+ }, z.core.$strict>;
9488
+ cases: z.ZodNumber;
9489
+ diverged: z.ZodNumber;
9490
+ refusedWrites: z.ZodNumber;
9491
+ errors: z.ZodNumber;
9492
+ stopped: z.ZodNumber;
9493
+ reads: z.ZodEnum<{
9494
+ recorded: "recorded";
9495
+ live: "live";
9496
+ }>;
9497
+ sampling: z.ZodObject<{
9498
+ models: z.ZodArray<z.ZodObject<{
9499
+ providerId: z.ZodString;
9500
+ model: z.ZodString;
9501
+ runs: z.ZodNumber;
9502
+ }, z.core.$strict>>;
9503
+ }, z.core.$strict>;
9504
+ repetitions: z.ZodNumber;
9505
+ metrics: z.ZodObject<{
9506
+ weightedYesShare: z.ZodObject<{
9507
+ baseline: z.ZodNullable<z.ZodNumber>;
9508
+ candidate: z.ZodNullable<z.ZodNumber>;
9509
+ delta: z.ZodNullable<z.ZodNumber>;
9510
+ n: z.ZodNumber;
9511
+ weight: z.ZodNumber;
9512
+ baselineN: z.ZodNumber;
9513
+ baselineWeight: z.ZodNumber;
9514
+ direction: z.ZodLiteral<"higher">;
9515
+ k: z.ZodOptional<z.ZodNumber>;
9516
+ spread: z.ZodOptional<z.ZodNumber>;
9517
+ }, z.core.$strict>;
9518
+ judgedCoverage: z.ZodObject<{
9519
+ baseline: z.ZodNullable<z.ZodNumber>;
9520
+ candidate: z.ZodNullable<z.ZodNumber>;
9521
+ delta: z.ZodNullable<z.ZodNumber>;
9522
+ n: z.ZodNumber;
9523
+ weight: z.ZodNumber;
9524
+ baselineN: z.ZodNumber;
9525
+ baselineWeight: z.ZodNumber;
9526
+ direction: z.ZodLiteral<"higher">;
9527
+ k: z.ZodOptional<z.ZodNumber>;
9528
+ spread: z.ZodOptional<z.ZodNumber>;
9529
+ }, z.core.$strict>;
9530
+ weightedPrecisionAtK: z.ZodObject<{
9531
+ baseline: z.ZodNullable<z.ZodNumber>;
9532
+ candidate: z.ZodNullable<z.ZodNumber>;
9533
+ delta: z.ZodNullable<z.ZodNumber>;
9534
+ n: z.ZodNumber;
9535
+ weight: z.ZodNumber;
9536
+ baselineN: z.ZodNumber;
9537
+ baselineWeight: z.ZodNumber;
9538
+ direction: z.ZodLiteral<"higher">;
9539
+ k: z.ZodOptional<z.ZodNumber>;
9540
+ spread: z.ZodOptional<z.ZodNumber>;
9541
+ }, z.core.$strict>;
9542
+ }, z.core.$strict>;
9543
+ }, z.core.$strict>;
9544
+ perCase: z.ZodArray<z.ZodObject<{
9545
+ caseId: z.ZodString;
9546
+ runIds: z.ZodArray<z.ZodString>;
9547
+ baseline: z.ZodObject<{
9548
+ yesWeight: z.ZodNumber;
9549
+ totalWeight: z.ZodNumber;
9550
+ items: z.ZodNumber;
9551
+ judgedItems: z.ZodNumber;
9552
+ topK: z.ZodObject<{
9553
+ yesWeight: z.ZodNumber;
9554
+ totalWeight: z.ZodNumber;
9555
+ }, z.core.$strict>;
9556
+ }, z.core.$strict>;
9557
+ candidate: z.ZodArray<z.ZodObject<{
9558
+ yesWeight: z.ZodNumber;
9559
+ totalWeight: z.ZodNumber;
9560
+ items: z.ZodNumber;
9561
+ judgedItems: z.ZodNumber;
9562
+ topK: z.ZodObject<{
9563
+ yesWeight: z.ZodNumber;
9564
+ totalWeight: z.ZodNumber;
9565
+ }, z.core.$strict>;
9566
+ }, z.core.$strict>>;
9567
+ changes: z.ZodOptional<z.ZodObject<{
9568
+ kept: z.ZodArray<z.ZodObject<{
9569
+ key: z.ZodString;
9570
+ rankBefore: z.ZodOptional<z.ZodNumber>;
9571
+ rank: z.ZodOptional<z.ZodNumber>;
9572
+ }, z.core.$strict>>;
9573
+ dropped: z.ZodArray<z.ZodObject<{
9574
+ key: z.ZodString;
9575
+ rankBefore: z.ZodOptional<z.ZodNumber>;
9576
+ }, z.core.$strict>>;
9577
+ new: z.ZodArray<z.ZodObject<{
9578
+ key: z.ZodString;
9579
+ pointer: z.ZodString;
9580
+ rank: z.ZodOptional<z.ZodNumber>;
9581
+ }, z.core.$strict>>;
9582
+ }, z.core.$strict>>;
9583
+ tools: z.ZodOptional<z.ZodArray<z.ZodObject<{
9584
+ step: z.ZodNumber;
9585
+ callId: z.ZodString;
9586
+ toolId: z.ZodString;
9587
+ toolVersion: z.ZodString;
9588
+ arguments: z.ZodUnknown;
9589
+ source: z.ZodEnum<{
9590
+ recorded: "recorded";
9591
+ live: "live";
9592
+ refused: "refused";
9593
+ }>;
9594
+ reason: z.ZodOptional<z.ZodString>;
9595
+ }, z.core.$strict>>>;
9596
+ diverged: z.ZodBoolean;
9597
+ refusedWrites: z.ZodNumber;
9598
+ noContext: z.ZodBoolean;
9599
+ approvalSkipped: z.ZodBoolean;
9600
+ error: z.ZodOptional<z.ZodString>;
9601
+ stopped: z.ZodOptional<z.ZodObject<{
9602
+ toolId: z.ZodString;
9603
+ arguments: z.ZodUnknown;
9604
+ reason: z.ZodOptional<z.ZodString>;
9605
+ }, z.core.$strict>>;
9606
+ }, z.core.$strict>>;
9607
+ }, z.core.$strict>;
7757
9608
  export type EvalRun = Schemas.EvalRun;
7758
9609
  declare const EvalRun: z.ZodObject<{
7759
9610
  runId: z.ZodUUID;
@@ -7767,6 +9618,7 @@ declare const EvalRun: z.ZodObject<{
7767
9618
  "human-review": "human-review";
7768
9619
  benchmark: "benchmark";
7769
9620
  custom: "custom";
9621
+ judged: "judged";
7770
9622
  }>;
7771
9623
  agentRef: z.ZodOptional<z.ZodObject<{
7772
9624
  agentId: z.ZodString;
@@ -7789,6 +9641,31 @@ declare const EvalRun: z.ZodObject<{
7789
9641
  result: z.ZodOptional<z.ZodRecord<z.ZodString, z.ZodUnknown>>;
7790
9642
  error: z.ZodOptional<z.ZodString>;
7791
9643
  correlationId: z.ZodOptional<z.ZodString>;
9644
+ comparison: z.ZodOptional<z.ZodObject<{
9645
+ baseline: z.ZodUnion<readonly [
9646
+ z.ZodLiteral<"recorded">,
9647
+ z.ZodObject<{
9648
+ agentId: z.ZodString;
9649
+ version: z.ZodString;
9650
+ }, z.core.$strict>,
9651
+ z.ZodObject<{
9652
+ live: z.ZodObject<{
9653
+ projectId: z.ZodOptional<z.ZodString>;
9654
+ segments: z.ZodOptional<z.ZodRecord<z.ZodString, z.ZodString>>;
9655
+ }, z.core.$strict>;
9656
+ }, z.core.$strict>
9657
+ ]>;
9658
+ reads: z.ZodEnum<{
9659
+ recorded: "recorded";
9660
+ live: "live";
9661
+ }>;
9662
+ repetitions: z.ZodNumber;
9663
+ k: z.ZodNumber;
9664
+ versions: z.ZodOptional<z.ZodObject<{
9665
+ tools: z.ZodOptional<z.ZodRecord<z.ZodString, z.ZodString>>;
9666
+ agents: z.ZodOptional<z.ZodRecord<z.ZodString, z.ZodString>>;
9667
+ }, z.core.$strict>>;
9668
+ }, z.core.$strict>>;
7792
9669
  }, z.core.$strict>;
7793
9670
  export type EvalRunCollectionPage = Schemas.EvalRunCollectionPage;
7794
9671
  declare const EvalRunCollectionPage: z.ZodObject<{
@@ -7804,6 +9681,7 @@ declare const EvalRunCollectionPage: z.ZodObject<{
7804
9681
  "human-review": "human-review";
7805
9682
  benchmark: "benchmark";
7806
9683
  custom: "custom";
9684
+ judged: "judged";
7807
9685
  }>;
7808
9686
  agentRef: z.ZodOptional<z.ZodObject<{
7809
9687
  agentId: z.ZodString;
@@ -7826,6 +9704,31 @@ declare const EvalRunCollectionPage: z.ZodObject<{
7826
9704
  result: z.ZodOptional<z.ZodRecord<z.ZodString, z.ZodUnknown>>;
7827
9705
  error: z.ZodOptional<z.ZodString>;
7828
9706
  correlationId: z.ZodOptional<z.ZodString>;
9707
+ comparison: z.ZodOptional<z.ZodObject<{
9708
+ baseline: z.ZodUnion<readonly [
9709
+ z.ZodLiteral<"recorded">,
9710
+ z.ZodObject<{
9711
+ agentId: z.ZodString;
9712
+ version: z.ZodString;
9713
+ }, z.core.$strict>,
9714
+ z.ZodObject<{
9715
+ live: z.ZodObject<{
9716
+ projectId: z.ZodOptional<z.ZodString>;
9717
+ segments: z.ZodOptional<z.ZodRecord<z.ZodString, z.ZodString>>;
9718
+ }, z.core.$strict>;
9719
+ }, z.core.$strict>
9720
+ ]>;
9721
+ reads: z.ZodEnum<{
9722
+ recorded: "recorded";
9723
+ live: "live";
9724
+ }>;
9725
+ repetitions: z.ZodNumber;
9726
+ k: z.ZodNumber;
9727
+ versions: z.ZodOptional<z.ZodObject<{
9728
+ tools: z.ZodOptional<z.ZodRecord<z.ZodString, z.ZodString>>;
9729
+ agents: z.ZodOptional<z.ZodRecord<z.ZodString, z.ZodString>>;
9730
+ }, z.core.$strict>>;
9731
+ }, z.core.$strict>>;
7829
9732
  }, z.core.$strict>>;
7830
9733
  nextCursor: z.ZodOptional<z.ZodString>;
7831
9734
  hasMore: z.ZodBoolean;
@@ -7843,6 +9746,29 @@ declare const StartEvalRunBody: z.ZodObject<{
7843
9746
  }, z.core.$strict>>;
7844
9747
  dryRun: z.ZodOptional<z.ZodBoolean>;
7845
9748
  correlationId: z.ZodOptional<z.ZodString>;
9749
+ baseline: z.ZodOptional<z.ZodUnion<readonly [
9750
+ z.ZodLiteral<"recorded">,
9751
+ z.ZodObject<{
9752
+ agentId: z.ZodString;
9753
+ version: z.ZodString;
9754
+ }, z.core.$strict>,
9755
+ z.ZodObject<{
9756
+ live: z.ZodObject<{
9757
+ projectId: z.ZodOptional<z.ZodString>;
9758
+ segments: z.ZodOptional<z.ZodRecord<z.ZodString, z.ZodString>>;
9759
+ }, z.core.$strict>;
9760
+ }, z.core.$strict>
9761
+ ]>>;
9762
+ reads: z.ZodOptional<z.ZodEnum<{
9763
+ recorded: "recorded";
9764
+ live: "live";
9765
+ }>>;
9766
+ repetitions: z.ZodOptional<z.ZodNumber>;
9767
+ k: z.ZodOptional<z.ZodNumber>;
9768
+ versions: z.ZodOptional<z.ZodObject<{
9769
+ tools: z.ZodOptional<z.ZodRecord<z.ZodString, z.ZodString>>;
9770
+ agents: z.ZodOptional<z.ZodRecord<z.ZodString, z.ZodString>>;
9771
+ }, z.core.$strict>>;
7846
9772
  }, z.core.$strict>;
7847
9773
  export type StartEvalRunResult = Schemas.StartEvalRunResult;
7848
9774
  declare const StartEvalRunResult: z.ZodObject<{
@@ -8041,11 +9967,47 @@ declare const DeploymentRecord: z.ZodObject<{
8041
9967
  }, z.core.$strict>>;
8042
9968
  agents: z.ZodArray<z.ZodObject<{
8043
9969
  id: z.ZodString;
8044
- version: z.ZodOptional<z.ZodString>;
9970
+ version: z.ZodString;
9971
+ authoredVersion: z.ZodOptional<z.ZodString>;
9972
+ reason: z.ZodOptional<z.ZodEnum<{
9973
+ "pins-changed": "pins-changed";
9974
+ unpinned: "unpinned";
9975
+ "version-taken": "version-taken";
9976
+ }>>;
9977
+ newVersion: z.ZodOptional<z.ZodBoolean>;
9978
+ pinChanges: z.ZodOptional<z.ZodArray<z.ZodObject<{
9979
+ kind: z.ZodEnum<{
9980
+ tool: "tool";
9981
+ prompt: "prompt";
9982
+ agent: "agent";
9983
+ setting: "setting";
9984
+ }>;
9985
+ id: z.ZodString;
9986
+ from: z.ZodOptional<z.ZodString>;
9987
+ to: z.ZodOptional<z.ZodString>;
9988
+ }, z.core.$strict>>>;
8045
9989
  }, z.core.$strict>>;
8046
9990
  flows: z.ZodArray<z.ZodObject<{
8047
9991
  id: z.ZodString;
8048
- version: z.ZodOptional<z.ZodString>;
9992
+ version: z.ZodString;
9993
+ authoredVersion: z.ZodOptional<z.ZodString>;
9994
+ reason: z.ZodOptional<z.ZodEnum<{
9995
+ "pins-changed": "pins-changed";
9996
+ unpinned: "unpinned";
9997
+ "version-taken": "version-taken";
9998
+ }>>;
9999
+ newVersion: z.ZodOptional<z.ZodBoolean>;
10000
+ pinChanges: z.ZodOptional<z.ZodArray<z.ZodObject<{
10001
+ kind: z.ZodEnum<{
10002
+ tool: "tool";
10003
+ prompt: "prompt";
10004
+ agent: "agent";
10005
+ setting: "setting";
10006
+ }>;
10007
+ id: z.ZodString;
10008
+ from: z.ZodOptional<z.ZodString>;
10009
+ to: z.ZodOptional<z.ZodString>;
10010
+ }, z.core.$strict>>>;
8049
10011
  }, z.core.$strict>>;
8050
10012
  }, z.core.$strict>;
8051
10013
  }, z.core.$strict>;
@@ -8090,11 +10052,47 @@ declare const DeploymentCollectionPage: z.ZodObject<{
8090
10052
  }, z.core.$strict>>;
8091
10053
  agents: z.ZodArray<z.ZodObject<{
8092
10054
  id: z.ZodString;
8093
- version: z.ZodOptional<z.ZodString>;
10055
+ version: z.ZodString;
10056
+ authoredVersion: z.ZodOptional<z.ZodString>;
10057
+ reason: z.ZodOptional<z.ZodEnum<{
10058
+ "pins-changed": "pins-changed";
10059
+ unpinned: "unpinned";
10060
+ "version-taken": "version-taken";
10061
+ }>>;
10062
+ newVersion: z.ZodOptional<z.ZodBoolean>;
10063
+ pinChanges: z.ZodOptional<z.ZodArray<z.ZodObject<{
10064
+ kind: z.ZodEnum<{
10065
+ tool: "tool";
10066
+ prompt: "prompt";
10067
+ agent: "agent";
10068
+ setting: "setting";
10069
+ }>;
10070
+ id: z.ZodString;
10071
+ from: z.ZodOptional<z.ZodString>;
10072
+ to: z.ZodOptional<z.ZodString>;
10073
+ }, z.core.$strict>>>;
8094
10074
  }, z.core.$strict>>;
8095
10075
  flows: z.ZodArray<z.ZodObject<{
8096
10076
  id: z.ZodString;
8097
- version: z.ZodOptional<z.ZodString>;
10077
+ version: z.ZodString;
10078
+ authoredVersion: z.ZodOptional<z.ZodString>;
10079
+ reason: z.ZodOptional<z.ZodEnum<{
10080
+ "pins-changed": "pins-changed";
10081
+ unpinned: "unpinned";
10082
+ "version-taken": "version-taken";
10083
+ }>>;
10084
+ newVersion: z.ZodOptional<z.ZodBoolean>;
10085
+ pinChanges: z.ZodOptional<z.ZodArray<z.ZodObject<{
10086
+ kind: z.ZodEnum<{
10087
+ tool: "tool";
10088
+ prompt: "prompt";
10089
+ agent: "agent";
10090
+ setting: "setting";
10091
+ }>;
10092
+ id: z.ZodString;
10093
+ from: z.ZodOptional<z.ZodString>;
10094
+ to: z.ZodOptional<z.ZodString>;
10095
+ }, z.core.$strict>>>;
8098
10096
  }, z.core.$strict>>;
8099
10097
  }, z.core.$strict>;
8100
10098
  }, z.core.$strict>>;
@@ -9046,7 +11044,17 @@ export interface AgentVersionsClient {
9046
11044
  * @wire POST /v1/agents/:agentId/versions/:version/unregister
9047
11045
  */
9048
11046
  unregister(agentId: AgentId, version: Semver, options?: MutationOptions): Promise<UnregisterAgentResult>;
11047
+ /**
11048
+ * Derive a new version from a pinned one with some data-block pins
11049
+ * swapped (an expert's edit, no code change). Numbered the next free
11050
+ * patch after the agent's highest version. Needs `publish` on the agent.
11051
+ *
11052
+ * @wire POST /v1/agents/:agentId/versions
11053
+ */
11054
+ derive(agentId: AgentId, input: DeriveAgentVersionInput, options?: MutationOptions): Promise<Agent$1>;
9049
11055
  }
11056
+ /** Body of `POST /v1/agents/{agentId}/versions`. */
11057
+ export type DeriveAgentVersionInput = DeriveAgentVersionBody;
9050
11058
  export interface ListAgentsFilter {
9051
11059
  readonly limit?: number;
9052
11060
  readonly cursor?: string;
@@ -9435,6 +11443,59 @@ export interface AuthProvidersClient {
9435
11443
  readonly idempotencyKey?: string;
9436
11444
  }): Promise<IdentityProviderUnregisterOutcome>;
9437
11445
  }
11446
+ export type BlockPage = BlockCollectionPage;
11447
+ /**
11448
+ * The block fields of `POST /v1/blocks`. The project travels as
11449
+ * `PublishBlockOptions.projectId`, so the input stays the block itself.
11450
+ */
11451
+ export type PublishBlockInput = Omit<PublishBlockBody, "projectId">;
11452
+ export interface PublishBlockOptions {
11453
+ /** The project the block belongs to. Its versions all stay there. */
11454
+ readonly projectId: string;
11455
+ readonly idempotencyKey?: string;
11456
+ }
11457
+ export interface ListBlocksFilter {
11458
+ readonly limit?: number;
11459
+ readonly cursor?: string;
11460
+ readonly kind?: BlockKind;
11461
+ /** Prefix match on the block id. */
11462
+ readonly name?: string;
11463
+ /** Only the blocks of this project, or of every project in this org. */
11464
+ readonly scope?: ScopeRef;
11465
+ }
11466
+ export interface ListBlockVersionsFilter {
11467
+ readonly limit?: number;
11468
+ readonly cursor?: string;
11469
+ /** Include unregistered versions, each with `unregisteredAt`. */
11470
+ readonly includeTombstoned?: boolean;
11471
+ }
11472
+ export interface BlocksClient {
11473
+ /**
11474
+ * Publish a block version into a project. Needs `write` on it.
11475
+ *
11476
+ * @wire POST /v1/blocks
11477
+ */
11478
+ publish(input: PublishBlockInput, options: PublishBlockOptions): Promise<PublishBlockResult>;
11479
+ /** @wire GET /v1/blocks */
11480
+ list(filter?: ListBlocksFilter): Promise<BlockPage>;
11481
+ /** @wire GET /v1/blocks/:blockId */
11482
+ get(blockId: string): Promise<Block>;
11483
+ readonly versions: BlockVersionsClient;
11484
+ }
11485
+ export interface BlockVersionsClient {
11486
+ /** @wire GET /v1/blocks/:blockId/versions */
11487
+ list(blockId: string, filter?: ListBlockVersionsFilter): Promise<BlockPage>;
11488
+ /** @wire GET /v1/blocks/:blockId/versions/:version */
11489
+ get(blockId: string, version: string): Promise<Block>;
11490
+ /** @wire POST /v1/blocks/:blockId/versions/:version/unregister */
11491
+ unregister(blockId: string, version: string, options?: {
11492
+ readonly idempotencyKey?: string;
11493
+ }): Promise<UnregisterBlockResult>;
11494
+ /** @wire POST /v1/blocks/:blockId/versions/:version/reinstate */
11495
+ reinstate(blockId: string, version: string, options?: {
11496
+ readonly idempotencyKey?: string;
11497
+ }): Promise<ReinstateBlockResult>;
11498
+ }
9438
11499
  /**
9439
11500
  * Capabilities resource — capability declarations + model router +
9440
11501
  * provider configuration.
@@ -9875,6 +11936,13 @@ export type EvalRunPage = EvalRunCollectionPage;
9875
11936
  */
9876
11937
  export type StartEvalRunInput = Omit<StartEvalRunBody, "projectId">;
9877
11938
  export type StartEvalRunOutcome = StartEvalRunResult;
11939
+ /**
11940
+ * A comparison's result, typed from the OpenAPI `JudgedComparisonResult`:
11941
+ * its summary and each case. `run` is a `judged` eval run (a test set
11942
+ * compared with a version); `undefined` for another kind of eval run, a
11943
+ * dry run, or one that hasn't finished.
11944
+ */
11945
+ export declare function comparisonOf(run: EvalRunRecord): JudgedComparisonResult | undefined;
9878
11946
  export interface StartEvalRunOptions {
9879
11947
  /** Project the eval run belongs to. The route requires it. */
9880
11948
  readonly projectId: string;
@@ -9927,6 +11995,16 @@ export type PublishSuiteInput = Omit<PublishEvalSuiteBody, "projectId">;
9927
11995
  export type PublishSuiteResult = PublishEvalSuiteResult;
9928
11996
  export type UnregisterSuiteVersionResult = UnregisterEvalSuiteResult;
9929
11997
  export type ReinstateSuiteVersionResult = ReinstateEvalSuiteVersionResult;
11998
+ /** Body of `POST /v1/eval-suites/{suiteId}/versions/from-judgments`. */
11999
+ export type BuildFromJudgmentsInput = BuildJudgedSuiteBody;
12000
+ export type BuildFromJudgmentsResult = BuildJudgedSuiteResult;
12001
+ /** One case of a `judged` suite: a copy of a judged run with its items' judgments summed up. */
12002
+ export type SuiteCase = JudgedEvalCase;
12003
+ export type SuiteCasePage = JudgedEvalCaseCollectionPage;
12004
+ export interface ListSuiteCasesQuery {
12005
+ readonly limit?: number;
12006
+ readonly cursor?: string;
12007
+ }
9930
12008
  export interface PublishSuiteOptions {
9931
12009
  /** Project the suite is published into. `POST /v1/eval-suites` requires it. */
9932
12010
  readonly projectId: string;
@@ -9958,6 +12036,22 @@ export interface EvalSuitesClient {
9958
12036
  list(filter?: ListSuitesFilter): Promise<SuitePage>;
9959
12037
  /** @wire GET /v1/eval-suites/:suiteId */
9960
12038
  get(suiteId: string): Promise<Suite>;
12039
+ /**
12040
+ * Publish a `judged` suite version whose cases are copies of judged runs
12041
+ * of an agent or flow, with each item's judgments summed up. Needs
12042
+ * `admin` on `input.projectId`.
12043
+ *
12044
+ * @wire POST /v1/eval-suites/:suiteId/versions/from-judgments
12045
+ */
12046
+ buildFromJudgments(suiteId: string, input: BuildFromJudgmentsInput, options?: {
12047
+ readonly idempotencyKey?: string;
12048
+ }): Promise<BuildFromJudgmentsResult>;
12049
+ /**
12050
+ * The cases of a `judged` suite version, in stored order.
12051
+ *
12052
+ * @wire GET /v1/eval-suites/:suiteId/versions/:version/cases
12053
+ */
12054
+ listCases(suiteId: string, version: string, query?: ListSuiteCasesQuery): Promise<SuiteCasePage>;
9961
12055
  readonly versions: EvalSuiteVersionsClient;
9962
12056
  }
9963
12057
  export interface EvalSuiteVersionsClient {
@@ -10290,6 +12384,114 @@ export interface IdentityUsersClient {
10290
12384
  readonly idempotencyKey?: string;
10291
12385
  }): Promise<RevokeSessionsOutcome>;
10292
12386
  }
12387
+ /**
12388
+ * Judge classes resource: the deployment's named kinds of judge
12389
+ * ("expert", "user", ...), each with a weight, scoped to the tenant, a
12390
+ * project, or an agent in a project.
12391
+ */
12392
+ export interface JudgeClassesClient {
12393
+ /**
12394
+ * Create a class. Names are unique among the live classes of a scope
12395
+ * (`409 judge-class-name-taken`).
12396
+ *
12397
+ * @wire `POST /v1/judge-classes` — see
12398
+ * `@kindgi/api/openapi.json#/paths/~1v1~1judge-classes/post`.
12399
+ */
12400
+ create(input: CreateJudgeClassInput, options?: {
12401
+ readonly idempotencyKey?: string;
12402
+ }): Promise<JudgeClass>;
12403
+ /**
12404
+ * Live classes, newest first. `scope` narrows to one scope.
12405
+ *
12406
+ * @wire `GET /v1/judge-classes` — see
12407
+ * `@kindgi/api/openapi.json#/paths/~1v1~1judge-classes/get`.
12408
+ */
12409
+ list(filter?: JudgeClassFilter): Promise<Page<JudgeClass>>;
12410
+ /**
12411
+ * One class, also a retired one (`unregisteredAt` set).
12412
+ *
12413
+ * @wire `GET /v1/judge-classes/{judgeClassId}` — see
12414
+ * `@kindgi/api/openapi.json#/paths/~1v1~1judge-classes~1{judgeClassId}/get`.
12415
+ */
12416
+ get(judgeClassId: string): Promise<JudgeClass>;
12417
+ /**
12418
+ * Change a live class's weight or description.
12419
+ *
12420
+ * @wire `PATCH /v1/judge-classes/{judgeClassId}` — see
12421
+ * `@kindgi/api/openapi.json#/paths/~1v1~1judge-classes~1{judgeClassId}/patch`.
12422
+ */
12423
+ update(judgeClassId: string, input: UpdateJudgeClassInput, options?: {
12424
+ readonly idempotencyKey?: string;
12425
+ }): Promise<JudgeClass>;
12426
+ /**
12427
+ * Retire a class: no new judgments may name it.
12428
+ *
12429
+ * @wire `POST /v1/judge-classes/{judgeClassId}/unregister` — see
12430
+ * `@kindgi/api/openapi.json#/paths/~1v1~1judge-classes~1{judgeClassId}~1unregister/post`.
12431
+ */
12432
+ unregister(judgeClassId: string, options?: {
12433
+ readonly idempotencyKey?: string;
12434
+ }): Promise<void>;
12435
+ }
12436
+ export interface JudgeClassFilter {
12437
+ readonly limit?: number;
12438
+ readonly cursor?: Cursor;
12439
+ readonly scope?: JudgeClassScope;
12440
+ }
12441
+ /**
12442
+ * Judgments resource: yes or no, with an optional reason, about one item
12443
+ * of a finished run's output.
12444
+ *
12445
+ * Who judged comes from the token you call with, never from the input.
12446
+ * Judging again as the same caller for the same run, item key and
12447
+ * `participantId` supersedes the earlier judgment, which stays as history.
12448
+ */
12449
+ export interface JudgmentsClient {
12450
+ /**
12451
+ * Judge an item of a finished run.
12452
+ *
12453
+ * @wire `POST /v1/judgments` — see `@kindgi/api/openapi.json#/paths/~1v1~1judgments/post`.
12454
+ */
12455
+ create(input: CreateJudgmentInput, options?: {
12456
+ readonly idempotencyKey?: string;
12457
+ }): Promise<Judgment>;
12458
+ /**
12459
+ * Live judgments, newest first.
12460
+ *
12461
+ * @wire `GET /v1/judgments` — see `@kindgi/api/openapi.json#/paths/~1v1~1judgments/get`.
12462
+ */
12463
+ list(filter?: JudgmentFilter): Promise<Page<Judgment>>;
12464
+ /**
12465
+ * One judgment with the stored copies of what was judged.
12466
+ *
12467
+ * @wire `GET /v1/judgments/{judgmentId}` — see
12468
+ * `@kindgi/api/openapi.json#/paths/~1v1~1judgments~1{judgmentId}/get`.
12469
+ */
12470
+ get(judgmentId: string): Promise<JudgmentWithCopies>;
12471
+ /**
12472
+ * Remove a judgment. Unknown ids fail with `404 judgment-not-found`.
12473
+ *
12474
+ * @wire `POST /v1/judgments/{judgmentId}/unregister` — see
12475
+ * `@kindgi/api/openapi.json#/paths/~1v1~1judgments~1{judgmentId}~1unregister/post`.
12476
+ */
12477
+ unregister(judgmentId: string, options?: {
12478
+ readonly idempotencyKey?: string;
12479
+ }): Promise<void>;
12480
+ }
12481
+ export interface JudgmentFilter {
12482
+ readonly limit?: number;
12483
+ readonly cursor?: Cursor;
12484
+ readonly runId?: string;
12485
+ readonly agentId?: string;
12486
+ /** Needs `agentId`. */
12487
+ readonly agentVersion?: string;
12488
+ readonly flowId?: string;
12489
+ readonly verdict?: Verdict;
12490
+ readonly judgeClassId?: string;
12491
+ readonly participantId?: string;
12492
+ /** Narrow to a project (or org). */
12493
+ readonly scope?: ScopeRef;
12494
+ }
10293
12495
  /**
10294
12496
  * MCP resource — Model Context Protocol interop.
10295
12497
  *
@@ -11104,6 +13306,10 @@ export interface ListRunsFilter {
11104
13306
  readonly topLevel?: boolean;
11105
13307
  /** Only this agent's turns, at any version (turns from before 0.1.3 don't name their agent). */
11106
13308
  readonly agentId?: AgentId | string;
13309
+ /** Replay runs (an eval run re-running a past run): `exclude` (the default) leaves them out, `include` lists them too, `only` lists just them. */
13310
+ readonly replays?: "exclude" | "include" | "only";
13311
+ /** Only the replay runs of this eval run (implies replays are included). */
13312
+ readonly evalRunId?: string;
11107
13313
  /** Include each run's `output` (omitted from lists by default). */
11108
13314
  readonly includeOutput?: boolean;
11109
13315
  }
@@ -12198,7 +14404,11 @@ export interface KindgiClient {
12198
14404
  readonly compliance: ComplianceClient;
12199
14405
  readonly audit: AuditResourceClient;
12200
14406
  readonly evalSuites: EvalSuitesClient;
14407
+ /** Data blocks: versioned prompts and settings an agent version pins. */
14408
+ readonly blocks: BlocksClient;
12201
14409
  readonly evalRuns: EvalRunsClient;
14410
+ readonly judgments: JudgmentsClient;
14411
+ readonly judgeClasses: JudgeClassesClient;
12202
14412
  readonly users: UsersClient;
12203
14413
  readonly identity: IdentityClient;
12204
14414
  readonly auth: AuthClient;