@datalayer/core 1.1.66 → 1.2.10

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (210) hide show
  1. package/README.md +41 -18
  2. package/lib/api/constants.d.ts +1 -1
  3. package/lib/api/constants.js +1 -1
  4. package/lib/api/contents/attachments.js +3 -7
  5. package/lib/api/contents/datasets.d.ts +48 -0
  6. package/lib/api/contents/datasets.js +44 -7
  7. package/lib/api/contents/runtimeMounts.js +1 -1
  8. package/lib/api/contents/sandboxUid.d.ts +13 -45
  9. package/lib/api/contents/sandboxUid.js +14 -65
  10. package/lib/api/evals/client.d.ts +402 -0
  11. package/lib/api/evals/client.js +359 -0
  12. package/lib/api/evals/index.d.ts +10 -0
  13. package/lib/api/evals/index.js +18 -0
  14. package/lib/api/evals/request.d.ts +43 -0
  15. package/lib/api/evals/request.js +67 -0
  16. package/lib/api/evals/status.d.ts +27 -0
  17. package/lib/api/evals/status.js +122 -0
  18. package/lib/api/evals/types.d.ts +1335 -0
  19. package/lib/api/evals/types.js +5 -0
  20. package/lib/api/iam/connectedAgents.d.ts +29 -0
  21. package/lib/api/iam/connectedAgents.js +35 -0
  22. package/lib/api/iam/identityProviders.d.ts +76 -0
  23. package/lib/api/iam/identityProviders.js +187 -0
  24. package/lib/api/iam/index.d.ts +1 -0
  25. package/lib/api/iam/index.js +1 -0
  26. package/lib/api/iam/mcpPolicy.d.ts +102 -1
  27. package/lib/api/iam/mcpPolicy.js +38 -2
  28. package/lib/api/iam/profile.d.ts +17 -1
  29. package/lib/api/iam/profile.js +27 -0
  30. package/lib/api/iam/scimTokens.d.ts +79 -0
  31. package/lib/api/iam/scimTokens.js +141 -0
  32. package/lib/api/iam/secrets.js +9 -21
  33. package/lib/api/iam/trials.d.ts +38 -0
  34. package/lib/api/iam/trials.js +66 -0
  35. package/lib/api/index.d.ts +4 -0
  36. package/lib/api/index.js +4 -0
  37. package/lib/api/mcp/observability.d.ts +25 -9
  38. package/lib/api/mcp/observability.js +63 -50
  39. package/lib/api/orchestration/client.d.ts +53 -0
  40. package/lib/api/orchestration/client.js +105 -0
  41. package/lib/api/orchestration/events.d.ts +88 -0
  42. package/lib/api/orchestration/events.js +245 -0
  43. package/lib/api/orchestration/generated.d.ts +693 -0
  44. package/lib/api/orchestration/generated.js +687 -0
  45. package/lib/api/orchestration/index.d.ts +30 -0
  46. package/lib/api/orchestration/index.js +34 -0
  47. package/lib/api/orchestration/lifecycle.d.ts +39 -0
  48. package/lib/api/orchestration/lifecycle.js +57 -0
  49. package/lib/api/orchestration/measures.d.ts +55 -0
  50. package/lib/api/orchestration/measures.js +99 -0
  51. package/lib/api/otel/dashboards.d.ts +56 -0
  52. package/lib/api/otel/dashboards.js +50 -0
  53. package/lib/api/otel/index.d.ts +1 -0
  54. package/lib/api/otel/index.js +1 -0
  55. package/lib/api/otel/metrics.d.ts +16 -0
  56. package/lib/api/otel/metrics.js +15 -0
  57. package/lib/api/scheduler/client.d.ts +32 -0
  58. package/lib/api/scheduler/client.js +42 -0
  59. package/lib/api/scheduler/index.d.ts +8 -0
  60. package/lib/api/scheduler/index.js +16 -0
  61. package/lib/api/scheduler/request.d.ts +28 -0
  62. package/lib/api/scheduler/request.js +50 -0
  63. package/lib/api/scheduler/types.d.ts +88 -0
  64. package/lib/api/scheduler/types.js +5 -0
  65. package/lib/api/spacer/comments.d.ts +97 -0
  66. package/lib/api/spacer/comments.js +45 -0
  67. package/lib/api/spacer/index.d.ts +9 -0
  68. package/lib/api/spacer/index.js +17 -0
  69. package/lib/api/spacer/notebooks.d.ts +33 -0
  70. package/lib/api/spacer/notebooks.js +39 -0
  71. package/lib/api/spacer/request.d.ts +25 -0
  72. package/lib/api/spacer/request.js +46 -0
  73. package/lib/api/spacer/spaces.d.ts +68 -0
  74. package/lib/api/spacer/spaces.js +45 -0
  75. package/lib/api/utils/validation.d.ts +4 -4
  76. package/lib/api/utils/validation.js +10 -7
  77. package/lib/client/constants.d.ts +1 -0
  78. package/lib/client/constants.js +1 -0
  79. package/lib/components/animation/AnimatedText.d.ts +1 -44
  80. package/lib/components/animation/AnimatedText.js +6 -122
  81. package/lib/components/anonymous/AnonymousKeyExpired.d.ts +56 -0
  82. package/lib/components/anonymous/AnonymousKeyExpired.js +96 -0
  83. package/lib/components/anonymous/AnonymousKeyTimer.d.ts +63 -0
  84. package/lib/components/anonymous/AnonymousKeyTimer.js +140 -0
  85. package/lib/components/anonymous/index.d.ts +7 -0
  86. package/lib/components/anonymous/index.js +15 -0
  87. package/lib/components/avatars/UserAvatar.d.ts +7 -1
  88. package/lib/components/avatars/UserAvatar.js +15 -1
  89. package/lib/components/billing/BillingEntitySelect.d.ts +6 -0
  90. package/lib/components/billing/BillingEntitySelect.js +49 -10
  91. package/lib/components/billing/eligibility.d.ts +22 -0
  92. package/lib/components/billing/eligibility.js +27 -0
  93. package/lib/components/billing/index.d.ts +1 -0
  94. package/lib/components/billing/index.js +1 -0
  95. package/lib/components/collaboration/LiveEditorCollaborators.d.ts +5 -0
  96. package/lib/components/collaboration/LiveEditorCollaborators.js +3 -1
  97. package/lib/components/display/DatalayerBox.d.ts +2 -3
  98. package/lib/components/display/DatalayerBox.js +1 -1
  99. package/lib/components/display/NavLink.js +3 -2
  100. package/lib/components/labels/StatusLabels.d.ts +2 -10
  101. package/lib/components/labels/StatusLabels.js +2 -26
  102. package/lib/components/principal/PrincipalAppearance.d.ts +21 -42
  103. package/lib/components/principal/PrincipalAppearance.js +232 -174
  104. package/lib/components/principal/PrincipalAvatar.d.ts +8 -1
  105. package/lib/components/principal/PrincipalAvatar.js +2 -2
  106. package/lib/components/principal/PrincipalDetailsOverlay.d.ts +8 -1
  107. package/lib/components/principal/PrincipalDetailsOverlay.js +2 -2
  108. package/lib/components/screencapture/Screencapture.js +46 -3
  109. package/lib/components/sharing/ShareAccessComponent.d.ts +45 -4
  110. package/lib/components/sharing/ShareAccessComponent.js +110 -78
  111. package/lib/components/sharing/SharingEditor.d.ts +5 -1
  112. package/lib/components/sharing/SharingEditor.js +9 -4
  113. package/lib/components/sharing/index.d.ts +1 -0
  114. package/lib/components/sharing/index.js +1 -0
  115. package/lib/components/sharing/sandboxSharing.d.ts +26 -0
  116. package/lib/components/sharing/sandboxSharing.js +41 -0
  117. package/lib/components/spaces/SpaceDetailsOverlay.d.ts +39 -0
  118. package/lib/components/spaces/SpaceDetailsOverlay.js +80 -0
  119. package/lib/components/spaces/SpaceDisplay.d.ts +38 -0
  120. package/lib/components/spaces/SpaceDisplay.js +92 -0
  121. package/lib/components/spaces/index.d.ts +4 -0
  122. package/lib/components/spaces/index.js +7 -0
  123. package/lib/components/spaces/spaceDisplayModel.d.ts +93 -0
  124. package/lib/components/spaces/spaceDisplayModel.js +170 -0
  125. package/lib/hooks/cacheConverters.d.ts +0 -7
  126. package/lib/hooks/cacheConverters.js +0 -25
  127. package/lib/hooks/index.d.ts +1 -0
  128. package/lib/hooks/index.js +1 -0
  129. package/lib/hooks/useBillingEntityStore.js +3 -0
  130. package/lib/hooks/useCache.d.ts +168 -25
  131. package/lib/hooks/useCache.js +1424 -307
  132. package/lib/hooks/useContents.d.ts +22 -1
  133. package/lib/hooks/useContents.js +90 -21
  134. package/lib/hooks/useMcp.d.ts +87 -1
  135. package/lib/hooks/useMcp.js +247 -12
  136. package/lib/hooks/useNavigate.d.ts +41 -3
  137. package/lib/hooks/useNavigate.js +106 -52
  138. package/lib/hooks/usePrincipalCacheStore.js +3 -0
  139. package/lib/hooks/usePrincipalStore.js +3 -0
  140. package/lib/hooks/useSpaceCacheStore.d.ts +50 -0
  141. package/lib/hooks/useSpaceCacheStore.js +85 -0
  142. package/lib/models/CreditsDTO.d.ts +6 -5
  143. package/lib/models/CreditsDTO.js +7 -5
  144. package/lib/models/McpBinding.d.ts +6 -0
  145. package/lib/models/Organization.d.ts +69 -0
  146. package/lib/models/Organization.js +96 -0
  147. package/lib/models/Secret.d.ts +1 -2
  148. package/lib/models/Secret.js +2 -18
  149. package/lib/models/StartedBy.d.ts +34 -0
  150. package/lib/models/StartedBy.js +36 -0
  151. package/lib/models/Team.d.ts +2 -0
  152. package/lib/models/Team.js +2 -0
  153. package/lib/models/User.d.ts +4 -0
  154. package/lib/models/User.js +3 -0
  155. package/lib/models/UserSettings.d.ts +13 -0
  156. package/lib/models/UserSettings.js +9 -0
  157. package/lib/models/contents/__fixtures__/v1-contracts.json +158 -30
  158. package/lib/navigation/components.js +2 -3
  159. package/lib/routes/publicPaths.js +4 -0
  160. package/lib/state/index.d.ts +1 -0
  161. package/lib/state/index.js +1 -0
  162. package/lib/state/sessionEnd.d.ts +55 -0
  163. package/lib/state/sessionEnd.js +162 -0
  164. package/lib/state/substates/CoreState.js +3 -3
  165. package/lib/state/substates/IAMState.js +7 -0
  166. package/lib/state/substates/LayoutState.js +12 -0
  167. package/lib/state/substates/ProfileState.d.ts +1 -0
  168. package/lib/state/substates/ProfileState.js +1 -0
  169. package/lib/utils/Lazy.d.ts +20 -0
  170. package/lib/utils/Lazy.js +32 -6
  171. package/lib/utils/Screencapture.js +3 -0
  172. package/lib/utils/WithSuspense.d.ts +11 -1
  173. package/lib/utils/WithSuspense.js +11 -1
  174. package/lib/views/mcp/AdmittedClients.d.ts +44 -0
  175. package/lib/views/mcp/AdmittedClients.js +59 -0
  176. package/lib/views/mcp/ApprovalQueue.d.ts +66 -0
  177. package/lib/views/mcp/ApprovalQueue.js +95 -0
  178. package/lib/views/mcp/ConnectedAgents.js +38 -3
  179. package/lib/views/mcp/EnterpriseConsole.d.ts +1 -1
  180. package/lib/views/mcp/EnterpriseConsole.js +23 -1
  181. package/lib/views/mcp/IdentityProviders.d.ts +33 -0
  182. package/lib/views/mcp/IdentityProviders.js +299 -0
  183. package/lib/views/mcp/McpDashboard.js +90 -7
  184. package/lib/views/mcp/McpObservability.d.ts +30 -1
  185. package/lib/views/mcp/McpObservability.js +105 -19
  186. package/lib/views/mcp/NotebookRuns.d.ts +39 -0
  187. package/lib/views/mcp/NotebookRuns.js +44 -0
  188. package/lib/views/mcp/OrganizationPolicy.js +2 -1
  189. package/lib/views/mcp/PersonalPolicy.js +2 -1
  190. package/lib/views/mcp/PolicyForm.d.ts +17 -1
  191. package/lib/views/mcp/PolicyForm.js +25 -3
  192. package/lib/views/mcp/PolicyHistory.d.ts +13 -0
  193. package/lib/views/mcp/PolicyHistory.js +14 -1
  194. package/lib/views/mcp/RunDetail.d.ts +87 -0
  195. package/lib/views/mcp/RunDetail.js +186 -0
  196. package/lib/views/mcp/ScimProvisioning.d.ts +54 -0
  197. package/lib/views/mcp/ScimProvisioning.js +173 -0
  198. package/lib/views/mcp/TeamPolicies.js +66 -5
  199. package/lib/views/mcp/ToolAccess.d.ts +60 -0
  200. package/lib/views/mcp/ToolAccess.js +101 -0
  201. package/lib/views/mcp/TraceTimeline.d.ts +81 -0
  202. package/lib/views/mcp/TraceTimeline.js +132 -0
  203. package/lib/views/mcp/index.d.ts +5 -0
  204. package/lib/views/mcp/index.js +5 -0
  205. package/lib/views/otel/simpleAuthStore.js +3 -0
  206. package/lib/views/secrets/SecretEdit.js +3 -2
  207. package/lib/views/secrets/SecretNew.js +5 -1
  208. package/package.json +269 -264
  209. package/scripts/generate-mcp-types.py +1 -1
  210. package/scripts/generate-orchestration-types.py +920 -0
@@ -0,0 +1,1335 @@
1
+ /**
2
+ * The evals objects as the AI Agents service sends them.
3
+ *
4
+ * These are the wire shapes, snake_case and all: the service's pydantic
5
+ * models (`EvalRecord`, `EvalExperimentRecord`, `EvalRunRecord`, …) are the
6
+ * source of truth and the landings UI reads these fields by their wire
7
+ * names. Nothing here is camel-cased on the way in — `metrics`, `summary`
8
+ * and `report` are free-form documents whose keys mean what the runner
9
+ * wrote, and a converter that renamed them would rename data.
10
+ *
11
+ * Product words: a **Benchmark** is an evalset, an **Experiment** is an
12
+ * experiment, a **Run** is a run, a **Task result** is a case result.
13
+ *
14
+ * @module api/evals/types
15
+ */
16
+ /** Where an evalset was created, which decides where its runs may execute. */
17
+ export type EvalRunEnvironment = 'ui' | 'sdk';
18
+ /** Fixed cases replayed (`batch`) or live events evaluated (`interactive`). */
19
+ export type EvalKind = 'batch' | 'interactive';
20
+ /** What a benchmark measures, the closed list section 12.5 renders by. */
21
+ export type EvalsetCategory = 'data' | 'coding' | 'tool-use' | 'model' | 'visual' | 'performance';
22
+ /** The dataset revision a benchmark reads; contents holds the revision. */
23
+ export interface DatasetRef {
24
+ source_uid: string;
25
+ revision_uid: string;
26
+ }
27
+ /** A named evaluator with its configuration, as a spec declares it. */
28
+ export interface EvalEvaluatorRef {
29
+ name: string;
30
+ config?: Record<string, unknown>;
31
+ [key: string]: unknown;
32
+ }
33
+ /** One case of an evalset: what is asked and what is expected. */
34
+ export interface EvalCase {
35
+ id: string;
36
+ name: string;
37
+ inputs: Record<string, unknown>;
38
+ expected_output: unknown;
39
+ evaluators: EvalEvaluatorRef[];
40
+ metadata: Record<string, unknown>;
41
+ }
42
+ /** An evalset, the benchmark: versioned cases, evaluators, metadata. */
43
+ export interface Evalset {
44
+ id: string;
45
+ owner_uid: string;
46
+ name: string;
47
+ description: string;
48
+ run_environment: EvalRunEnvironment;
49
+ kind: EvalKind;
50
+ category: EvalsetCategory | '';
51
+ dataset_ref: DatasetRef | null;
52
+ /** Moves on every change to cases, evaluators, schema or category. */
53
+ version: number;
54
+ /** How many launches this benchmark has had; the next is number + 1. */
55
+ launch_count: number;
56
+ /** What the runs say, kept by the launch refresh (B2-10). */
57
+ run_count?: number;
58
+ latest_launch_id?: string;
59
+ latest_pass_rate?: number | null;
60
+ best_pass_rate?: number | null;
61
+ last_run_status?: string;
62
+ last_run_at?: string | null;
63
+ estimated_cost?: number | null;
64
+ environment?: string;
65
+ subject_refs?: string[];
66
+ schema: Record<string, unknown>;
67
+ evalset_evaluators: EvalEvaluatorRef[];
68
+ report_evaluators: EvalEvaluatorRef[];
69
+ tags: string[];
70
+ metadata: Record<string, unknown>;
71
+ cases: EvalCase[];
72
+ is_public: boolean;
73
+ /** The benchmark this one was taken from, and its version then (B5-03). */
74
+ derived_from_uid?: string;
75
+ derived_from_version?: number | null;
76
+ created_at: string;
77
+ updated_at: string;
78
+ archived: boolean;
79
+ }
80
+ /** One version of a benchmark's definition, kept as it was. */
81
+ export interface EvalsetVersion {
82
+ evalset_id: string;
83
+ version: number;
84
+ name: string;
85
+ category: string;
86
+ schema: Record<string, unknown>;
87
+ evalset_evaluators: EvalEvaluatorRef[];
88
+ report_evaluators: EvalEvaluatorRef[];
89
+ cases: EvalCase[];
90
+ created_at: string | null;
91
+ }
92
+ /** One experiment: an agentspec, model, prompt or configuration under test. */
93
+ /** What an experiment runs (B2-11): `{kind, ref, model?, ...}`. */
94
+ export interface EvalSubject {
95
+ kind: string;
96
+ ref: string;
97
+ model?: string;
98
+ endpoint?: string;
99
+ image?: string;
100
+ command?: string;
101
+ provider?: string;
102
+ }
103
+ export interface EvalExperiment {
104
+ id: string;
105
+ owner_uid: string;
106
+ evalset_id: string | null;
107
+ name: string;
108
+ description: string;
109
+ status: string;
110
+ /** What the experiment runs (B2-11); empty on an experiment without one. */
111
+ subject?: EvalSubject | Record<string, never>;
112
+ config: Record<string, unknown>;
113
+ summary: Record<string, unknown>;
114
+ tags: string[];
115
+ created_at: string;
116
+ updated_at: string;
117
+ archived: boolean;
118
+ }
119
+ /** A case's outcome inside a run's metrics. */
120
+ export interface EvalCaseResult {
121
+ name: string;
122
+ passed: boolean;
123
+ status: string;
124
+ score: number;
125
+ category?: string;
126
+ difficulty?: string;
127
+ prompt?: string;
128
+ output?: string;
129
+ [key: string]: unknown;
130
+ }
131
+ /** An evaluator's aggregate inside a run's metrics. */
132
+ export interface EvalEvaluatorResult {
133
+ name: string;
134
+ scope: 'evalset' | 'report' | string;
135
+ score?: number | null;
136
+ passed?: boolean | null;
137
+ passed_cases?: number;
138
+ total_cases?: number;
139
+ summary?: string;
140
+ threshold?: number;
141
+ observed?: number;
142
+ [key: string]: unknown;
143
+ }
144
+ /** What the runner measured for a run. Free-form beyond the known keys. */
145
+ export interface EvalRunMetrics {
146
+ pass_rate?: number;
147
+ total_cases?: number;
148
+ passed?: number;
149
+ failed?: number;
150
+ avg_score?: number;
151
+ case_results?: EvalCaseResult[];
152
+ evaluator_results?: EvalEvaluatorResult[];
153
+ [key: string]: unknown;
154
+ }
155
+ /** One execution of an experiment, inside a launch. */
156
+ export interface EvalRun {
157
+ id: string;
158
+ experiment_id: string;
159
+ /** The evalset the experiment belongs to; empty on rows not yet backfilled. */
160
+ evalset_id: string;
161
+ /** The launch this run is part of; every run has one once backfilled. */
162
+ launch_id: string;
163
+ /** Credits consumed so far (B2-06). */
164
+ cost_credits?: number | null;
165
+ /** The credits this run may spend; null is no cap (B2-06). */
166
+ budget_limit?: number | null;
167
+ /** The compute asked for: environment, slots, concurrency, time_reservation (B2-06). */
168
+ compute?: Record<string, unknown>;
169
+ elapsed_ms?: number | null;
170
+ /** Why the run is `blocked`, when it is (B2-06). */
171
+ blocked_reason?: string;
172
+ /** The durable workflow executing it, once one does. */
173
+ workflow_uid: string;
174
+ /** The evalset version the run was started against; null on rows not yet backfilled. */
175
+ evalset_version?: number | null;
176
+ owner_uid: string;
177
+ status: string;
178
+ started_at: string | null;
179
+ ended_at: string | null;
180
+ metrics: EvalRunMetrics;
181
+ summary: Record<string, unknown>;
182
+ report: Record<string, unknown>;
183
+ /** Cases done, running and failed so far, as the workflow projects them. */
184
+ progress: Record<string, unknown>;
185
+ created_at: string;
186
+ updated_at: string;
187
+ }
188
+ /** A task result: one case of one run, its own document. */
189
+ export interface EvalTaskResult {
190
+ id: string;
191
+ run_id: string;
192
+ launch_id: string;
193
+ experiment_id: string;
194
+ evalset_id: string;
195
+ owner_uid: string;
196
+ case_id: string;
197
+ name: string;
198
+ status: string;
199
+ score: number | null;
200
+ passed: boolean | null;
201
+ explanation: string;
202
+ category: string;
203
+ difficulty: string;
204
+ failure_mode: string;
205
+ failure_stage: string;
206
+ prompt: string;
207
+ output: string;
208
+ trajectory: unknown[];
209
+ sandbox_snapshot_uid: string;
210
+ notebook_uid: string;
211
+ /** The investigation opened on this task, once one is (B3-02). */
212
+ investigation_uid?: string;
213
+ /**
214
+ * The root execution this task was delegated as — a team subject only; empty
215
+ * for an agentspec or a model, which are one call and no tree. Absent from a
216
+ * service older than the field.
217
+ */
218
+ execution_id?: string;
219
+ /** The trace that execution runs under (the W3C trace id), when there is one. */
220
+ trace_id?: string;
221
+ artifact_uids: string[];
222
+ cost_credits: number | null;
223
+ duration_ms: number | null;
224
+ reviewed_by: string;
225
+ started_at: string | null;
226
+ ended_at: string | null;
227
+ created_at: string | null;
228
+ updated_at: string | null;
229
+ }
230
+ /** A launch: one submission of a benchmark across its experiments. */
231
+ export interface EvalLaunch {
232
+ id: string;
233
+ owner_uid: string;
234
+ account_uid: string;
235
+ evalset_id: string;
236
+ evalset_version: number | null;
237
+ /** The number people say: "Run 128". */
238
+ number: number | null;
239
+ experiment_ids: string[];
240
+ run_ids: string[];
241
+ status: string;
242
+ run_mode: EvalKind;
243
+ config: Record<string, unknown>;
244
+ progress: Record<string, unknown>;
245
+ created_by_uid: string;
246
+ /** The report document written for this launch, once one is (B3-03). */
247
+ report_document_uid?: string;
248
+ /** The credits the launch may spend; null is no cap (B2-06). */
249
+ budget_limit?: number | null;
250
+ /** The compute asked for: environment, slots, concurrency, time_reservation (B2-06). */
251
+ compute?: Record<string, unknown>;
252
+ /** Why the launch is `blocked`, when it is (B2-06). */
253
+ blocked_reason?: string;
254
+ /** The launch this one runs again, at the benchmark version it ran (B5-06). */
255
+ reproduces_launch_id?: string;
256
+ /**
257
+ * The window of live traffic an interactive launch is (B2-13):
258
+ * `{size, starts_at, ends_at}`. Empty on a batch launch.
259
+ */
260
+ window?: {
261
+ size?: string;
262
+ starts_at?: string;
263
+ ends_at?: string;
264
+ };
265
+ archived: boolean;
266
+ started_at: string | null;
267
+ ended_at: string | null;
268
+ created_at: string | null;
269
+ updated_at: string | null;
270
+ }
271
+ /** One problem the pre-launch validation found (B2-07). */
272
+ export interface LaunchPlanProblem {
273
+ code: string;
274
+ severity: 'error' | 'warning';
275
+ message: string;
276
+ experiment_id?: string;
277
+ }
278
+ /** The plan of a launch before it is made (B2-07). */
279
+ export interface LaunchPlanResponse {
280
+ success: boolean;
281
+ /** No error-level problem: the launch would run. */
282
+ ok: boolean;
283
+ /** The mode the plan was made for (B2-13). */
284
+ run_mode: EvalKind;
285
+ /**
286
+ * How run mode, run environment and execution target relate, stated by
287
+ * the service so the wizard explains it inline rather than failing after
288
+ * submission (BENCHMARK.md section 9).
289
+ */
290
+ run_mode_constraint: string;
291
+ problems: LaunchPlanProblem[];
292
+ estimate: {
293
+ cases: number;
294
+ experiments: number;
295
+ slots: number;
296
+ duration_seconds: number;
297
+ duration_basis: 'history' | 'assumed' | 'mixed';
298
+ /** Credits the launch would reserve; null when the burning rate is unknown. */
299
+ credits_reserved: number | null;
300
+ credits_available: number | null;
301
+ budget_limit: number | null;
302
+ };
303
+ compute: {
304
+ environment: string;
305
+ slots: number;
306
+ concurrency: number;
307
+ time_reservation: number;
308
+ burning_rate: number | null;
309
+ };
310
+ experiments: Array<{
311
+ id: string;
312
+ name: string;
313
+ subject: EvalSubject | Record<string, never>;
314
+ execution_target: string;
315
+ seconds_per_case: number;
316
+ duration_basis: 'history' | 'assumed';
317
+ duration_seconds: number;
318
+ credits_reserved: number | null;
319
+ }>;
320
+ approvals: {
321
+ pending_tool_approvals: number;
322
+ budget_limit: number | null;
323
+ };
324
+ unsupported_evaluators: string[];
325
+ }
326
+ /** What a claim moved, by document type; empty when it was a repeat (B2-14). */
327
+ export interface ClaimTrialResponse {
328
+ success: boolean;
329
+ trial: Record<string, unknown> | null;
330
+ moved: Record<string, number>;
331
+ }
332
+ /** The subjects an experiment can have and the models offered (B2-11). */
333
+ export interface SubjectsResponse {
334
+ success: boolean;
335
+ kinds: Array<{
336
+ kind: string;
337
+ executable: boolean;
338
+ label: string;
339
+ }>;
340
+ models: string[];
341
+ default_model: string;
342
+ default_provider: string;
343
+ /** False when AI Inference could not be asked. */
344
+ models_available: boolean;
345
+ }
346
+ /** One evaluated event of a live target. */
347
+ export interface LiveEvalEvent {
348
+ id: string;
349
+ owner_uid: string;
350
+ target_id: string;
351
+ target_type: string;
352
+ evaluator_name: string;
353
+ metric_name: string;
354
+ value_num: number | null;
355
+ label: string;
356
+ passed: boolean | null;
357
+ attributes: Record<string, unknown>;
358
+ /** The experiment the event was produced for, graded with its evaluators (B2-13). */
359
+ experiment_id?: string;
360
+ evalset_id?: string;
361
+ /** The interactive launch whose window the event fell into, if one was open. */
362
+ launch_id?: string;
363
+ /** The case it answered, when it answered one. */
364
+ case_id?: string;
365
+ /** What the experiment's evaluators said of it, server-side. */
366
+ results?: Array<Record<string, unknown>>;
367
+ created_at: string;
368
+ }
369
+ /** What the rolling windows raised on a live target (B2-13). */
370
+ export interface LiveEvalAlert {
371
+ id: string;
372
+ owner_uid: string;
373
+ target_id: string;
374
+ target_type: string;
375
+ experiment_id: string;
376
+ launch_id: string;
377
+ /** `failure_spike` or `drift`. */
378
+ kind: string;
379
+ message: string;
380
+ /** The failed share, or the pass rate that drifted. */
381
+ value: number | null;
382
+ /** The pass rate before the drift. */
383
+ baseline: number | null;
384
+ threshold: number | null;
385
+ window_events: number;
386
+ created_at: string | null;
387
+ }
388
+ /** A live target's rolling-window summary. */
389
+ export interface LiveEvalTarget {
390
+ target_id: string;
391
+ target_type: string;
392
+ event_count: number;
393
+ passed_count: number;
394
+ failed_count?: number;
395
+ pass_rate: number | null;
396
+ avg_value: number | null;
397
+ last_event_at: string | null;
398
+ /** What the target's newest event is bound to (B2-13). */
399
+ experiment_id?: string;
400
+ evalset_id?: string;
401
+ launch_id?: string;
402
+ [key: string]: unknown;
403
+ }
404
+ export interface CreateEvalsetRequest {
405
+ name: string;
406
+ description?: string;
407
+ run_environment?: EvalRunEnvironment;
408
+ kind?: EvalKind;
409
+ category?: EvalsetCategory;
410
+ dataset_ref?: DatasetRef;
411
+ schema?: Record<string, unknown>;
412
+ evalset_evaluators?: EvalEvaluatorRef[];
413
+ report_evaluators?: EvalEvaluatorRef[];
414
+ tags?: string[];
415
+ metadata?: Record<string, unknown>;
416
+ cases?: Array<Partial<EvalCase>>;
417
+ is_public?: boolean;
418
+ }
419
+ /** Every field optional; `kind` is not updatable through the API. */
420
+ export type UpdateEvalsetRequest = Partial<Omit<CreateEvalsetRequest, 'kind'>>;
421
+ export interface CaseRequest {
422
+ name: string;
423
+ inputs?: Record<string, unknown>;
424
+ expected_output?: unknown;
425
+ evaluators?: EvalEvaluatorRef[];
426
+ metadata?: Record<string, unknown>;
427
+ }
428
+ export interface CreateExperimentRequest {
429
+ evalset_id?: string | null;
430
+ name: string;
431
+ description?: string;
432
+ status?: string;
433
+ config?: Record<string, unknown>;
434
+ summary?: Record<string, unknown>;
435
+ tags?: string[];
436
+ }
437
+ export type UpdateExperimentRequest = Partial<CreateExperimentRequest>;
438
+ export interface CreateRunRequest {
439
+ status?: string;
440
+ started_at?: string | null;
441
+ ended_at?: string | null;
442
+ metrics?: Record<string, unknown>;
443
+ summary?: Record<string, unknown>;
444
+ report?: Record<string, unknown>;
445
+ }
446
+ export interface CreateLiveEventRequest {
447
+ target_id: string;
448
+ target_type?: string;
449
+ evaluator_name: string;
450
+ metric_name: string;
451
+ value_num?: number | null;
452
+ label?: string;
453
+ passed?: boolean | null;
454
+ attributes?: Record<string, unknown>;
455
+ created_at?: string | null;
456
+ }
457
+ export interface EvalsetResponse {
458
+ success: boolean;
459
+ evalset: Evalset;
460
+ }
461
+ /** One thing a published package holds, as the review enumerates it (B5-02). */
462
+ export interface EvalsetPackageItem {
463
+ /** Which part of the package it belongs to: definition, results, report… */
464
+ section: string;
465
+ kind: string;
466
+ uid: string;
467
+ label: string;
468
+ }
469
+ /** What stands in the way of a publication (section 14.5). */
470
+ export interface EvalsetPackageProblem {
471
+ code: string;
472
+ detail: string;
473
+ uid?: string;
474
+ }
475
+ /** What a package leaves out, and why: nothing is published by implication. */
476
+ export interface EvalsetPackageExclusion {
477
+ kind: string;
478
+ uid: string;
479
+ reason: string;
480
+ }
481
+ /**
482
+ * The package a published benchmark is: a snapshot of everything the numbers
483
+ * depend on, what would be published, and what refuses it.
484
+ */
485
+ export interface EvalsetPackage {
486
+ contents: Record<string, unknown>;
487
+ items: EvalsetPackageItem[];
488
+ problems: EvalsetPackageProblem[];
489
+ excluded: EvalsetPackageExclusion[];
490
+ checksum: string;
491
+ }
492
+ /** A package as it was written: immutable, and reversibly visible. */
493
+ export interface EvalsetPublication {
494
+ id: string;
495
+ evalset_id: string;
496
+ evalset_version: number | null;
497
+ owner_uid: string;
498
+ actor_uid?: string;
499
+ report_id: string;
500
+ launch_ids: string[];
501
+ contents: Record<string, unknown>;
502
+ items: EvalsetPackageItem[];
503
+ checksum: string;
504
+ status: string;
505
+ note: string;
506
+ created_at?: string | null;
507
+ unpublished_at?: string | null;
508
+ }
509
+ /** One launch in a comparison of launches of the same benchmark (B5-08). */
510
+ export interface EvalsetComparisonLaunch {
511
+ launch_id: string;
512
+ number: number | null;
513
+ status: string;
514
+ run_mode: string;
515
+ evalset_version: number | null;
516
+ started_at?: string | null;
517
+ ended_at?: string | null;
518
+ created_at?: string | null;
519
+ run_ids: string[];
520
+ pass_rate: number | null;
521
+ cost_credits: number | null;
522
+ elapsed_ms: number | null;
523
+ reproduces_launch_id: string;
524
+ }
525
+ /** What one subject scored in each of the launches compared. */
526
+ export interface EvalsetComparisonSubject {
527
+ ref: string;
528
+ label: string;
529
+ kind: string;
530
+ pass_rates: Record<string, number | null>;
531
+ /** The last launch against the first; `null` where either has no rate. */
532
+ delta: number | null;
533
+ }
534
+ /** What one task did in each launch: what broke, and what was fixed. */
535
+ export interface EvalsetComparisonTask {
536
+ name: string;
537
+ statuses: Record<string, string>;
538
+ scores: Record<string, number>;
539
+ regressed: boolean;
540
+ fixed: boolean;
541
+ }
542
+ export interface EvalsetComparisonDelta {
543
+ earlier_launch_id: string;
544
+ earlier: number | null;
545
+ later_launch_id: string;
546
+ later: number | null;
547
+ delta: number;
548
+ }
549
+ export interface EvalsetComparison {
550
+ launches: EvalsetComparisonLaunch[];
551
+ subjects: EvalsetComparisonSubject[];
552
+ tasks: EvalsetComparisonTask[];
553
+ deltas: EvalsetComparisonDelta[];
554
+ /** How a delta is read, stated rather than assumed. */
555
+ convention: string;
556
+ regressions: number;
557
+ fixes: number;
558
+ }
559
+ /** One place on a published benchmark's leaderboard (B5-08). */
560
+ export interface EvalsetLeaderboardRow {
561
+ place: number;
562
+ ref: string;
563
+ label: string;
564
+ kind: string;
565
+ runs: number;
566
+ best_pass_rate: number | null;
567
+ latest_pass_rate: number | null;
568
+ best_run_id: string;
569
+ /** The package the best run was published in, so a reader can open it. */
570
+ publication_id: string;
571
+ evalset_version: number | null;
572
+ }
573
+ export interface EvalsetLeaderboard {
574
+ rows: EvalsetLeaderboardRow[];
575
+ /** Runs a package published; only these are counted. */
576
+ published_runs: number;
577
+ counted_runs: number;
578
+ }
579
+ /** A benchmark that has run one subject, with what its runs scored (B5-12). */
580
+ export interface SubjectBenchmark {
581
+ evalset_id: string;
582
+ name: string;
583
+ category: string;
584
+ is_public: boolean;
585
+ runs: number;
586
+ latest_pass_rate: number | null;
587
+ best_pass_rate: number | null;
588
+ last_run_at: string | null;
589
+ cost_credits: number;
590
+ }
591
+ /** One run of a subject, as its page lists it. */
592
+ export interface SubjectRun {
593
+ run_id: string;
594
+ evalset_id: string;
595
+ evalset_name: string;
596
+ launch_id: string;
597
+ experiment_id: string;
598
+ status: string;
599
+ pass_rate: number | null;
600
+ cost_credits: number | null;
601
+ elapsed_ms: number | null;
602
+ evalset_version: number | null;
603
+ ended_at?: string | null;
604
+ created_at?: string | null;
605
+ }
606
+ export interface SubjectScore {
607
+ at: string | null;
608
+ pass_rate: number;
609
+ evalset_id: string;
610
+ run_id: string;
611
+ }
612
+ export interface SubjectLaunch {
613
+ launch_id: string;
614
+ evalset_id: string;
615
+ number: number | null;
616
+ status: string;
617
+ }
618
+ export interface SubjectTotals {
619
+ benchmarks: number;
620
+ runs: number;
621
+ cost_credits: number | null;
622
+ elapsed_ms: number | null;
623
+ latest_pass_rate: number | null;
624
+ best_pass_rate: number | null;
625
+ active_launches: number;
626
+ }
627
+ /**
628
+ * What an agent or a model has actually run. Empty where it has never been
629
+ * run: the page says so rather than drawing a history nobody ran.
630
+ */
631
+ export interface SubjectUsage {
632
+ subject_ref: string;
633
+ benchmarks: SubjectBenchmark[];
634
+ runs: SubjectRun[];
635
+ scores: SubjectScore[];
636
+ active_launches: SubjectLaunch[];
637
+ totals: SubjectTotals;
638
+ }
639
+ /** What a search made of the words somebody typed (B5-11, section 19). */
640
+ export interface EvalsSearchQuery {
641
+ raw: string;
642
+ /** What is left to look for by name, once the words that mean something else are taken out. */
643
+ text: string;
644
+ /** A status word, as the records spell it: `failed`, `in_review`… */
645
+ status: string;
646
+ /** `run 128` names a launch. */
647
+ launch_number: number | null;
648
+ /** A dataset revision somebody named, such as `v2026.09`. */
649
+ revision: string;
650
+ }
651
+ export type EvalsSearchKind = 'benchmark' | 'launch' | 'task' | 'report' | 'investigation';
652
+ /** One thing found, keeping its kind and what it belongs to. */
653
+ export interface EvalsSearchRow {
654
+ kind: EvalsSearchKind | string;
655
+ uid: string;
656
+ title: string;
657
+ subtitle: string;
658
+ status: string;
659
+ score?: number | null;
660
+ is_public?: boolean;
661
+ evalset_id?: string;
662
+ launch_id?: string;
663
+ run_id?: string;
664
+ case_id?: string;
665
+ at?: string | null;
666
+ }
667
+ export interface EvalsSearchResponse {
668
+ success: boolean;
669
+ query: EvalsSearchQuery;
670
+ results: EvalsSearchRow[];
671
+ counts: Record<string, number>;
672
+ total: number;
673
+ }
674
+ export interface SubjectUsageResponse {
675
+ success: boolean;
676
+ usage: SubjectUsage;
677
+ }
678
+ /**
679
+ * What benchmark work a sandbox is doing (B5-12, section 15.4).
680
+ *
681
+ * A pool slot's sandbox is named after the run it serves, and a sandbox
682
+ * restored from a task's snapshot is found by that snapshot; either way the
683
+ * page can say which benchmark, launch, run and task the machine belongs to.
684
+ * `found` is false for a sandbox somebody opened themselves, which is the
685
+ * ordinary case and not an error.
686
+ */
687
+ export interface SandboxProvenance {
688
+ found: boolean;
689
+ benchmark: {
690
+ id: string;
691
+ name: string;
692
+ } | null;
693
+ launch: {
694
+ id: string;
695
+ number: number | null;
696
+ status: string;
697
+ } | null;
698
+ run: {
699
+ id: string;
700
+ status: string;
701
+ experiment_id: string;
702
+ experiment_name: string;
703
+ /** Which of the pool's sandboxes this one is, where it is one. */
704
+ slot: number | null;
705
+ } | null;
706
+ task: {
707
+ case_id: string;
708
+ name: string;
709
+ status: string;
710
+ snapshot_uid: string;
711
+ } | null;
712
+ /** The investigation a restored sandbox was brought back for (B3-05). */
713
+ investigation: {
714
+ id: string;
715
+ title: string;
716
+ status: string;
717
+ } | null;
718
+ /** Paths of the app. */
719
+ links: Partial<Record<'benchmark' | 'launch' | 'run' | 'task' | 'investigation', string>>;
720
+ }
721
+ export interface SandboxProvenanceResponse {
722
+ success: boolean;
723
+ provenance: SandboxProvenance;
724
+ }
725
+ /**
726
+ * Whether anything will run the account's benchmarks (B5-12, section 15.2).
727
+ *
728
+ * A launch at `queued` looks the same whether the queue is busy or nothing
729
+ * polls it. `serving` is the difference. The depth is the account's own
730
+ * launches: the executor's other work belongs to nobody on this page.
731
+ */
732
+ export interface BenchmarkCompute {
733
+ serving: boolean;
734
+ engine: string;
735
+ queue: string;
736
+ queue_served: boolean;
737
+ application_version: string;
738
+ /** Why nothing is serving, when nothing is. */
739
+ detail: string;
740
+ queued: number;
741
+ running: number;
742
+ /** The slots the running launches hold. */
743
+ sandboxes: number;
744
+ }
745
+ export interface BenchmarkComputeResponse {
746
+ success: boolean;
747
+ compute: BenchmarkCompute;
748
+ }
749
+ export interface EvalsetComparisonResponse {
750
+ success: boolean;
751
+ comparison: EvalsetComparison;
752
+ }
753
+ export interface EvalsetLeaderboardResponse {
754
+ success: boolean;
755
+ leaderboard: EvalsetLeaderboard;
756
+ }
757
+ export interface EvalsetPackagePreviewResponse {
758
+ success: boolean;
759
+ package: EvalsetPackage;
760
+ }
761
+ export interface EvalsetPublicationResponse {
762
+ success: boolean;
763
+ publication: EvalsetPublication;
764
+ }
765
+ export interface EvalsetPublicationListResponse {
766
+ success: boolean;
767
+ total: number;
768
+ publications: EvalsetPublication[];
769
+ }
770
+ export interface EvalsetListResponse {
771
+ success: boolean;
772
+ total: number;
773
+ evalsets: Evalset[];
774
+ }
775
+ export interface EvalsetDeleteResponse {
776
+ success: boolean;
777
+ cascade: {
778
+ experiments_deleted: number;
779
+ runs_deleted: number;
780
+ cases_deleted: number;
781
+ };
782
+ }
783
+ export interface CaseListResponse {
784
+ success: boolean;
785
+ total: number;
786
+ cases: EvalCase[];
787
+ }
788
+ export interface CaseResponse {
789
+ success: boolean;
790
+ case: EvalCase;
791
+ }
792
+ export interface PublicEvalsetDetailsResponse {
793
+ success: boolean;
794
+ evalset: Evalset;
795
+ cases: EvalCase[];
796
+ experiments: EvalExperiment[];
797
+ experiments_total: number;
798
+ runs_limit: number;
799
+ runs_by_experiment: Record<string, EvalRun[]>;
800
+ runs_total_by_experiment: Record<string, number>;
801
+ }
802
+ export interface ExperimentResponse {
803
+ success: boolean;
804
+ experiment: EvalExperiment;
805
+ }
806
+ export interface ExperimentListResponse {
807
+ success: boolean;
808
+ total: number;
809
+ experiments: EvalExperiment[];
810
+ }
811
+ export interface ExperimentDeleteResponse {
812
+ success: boolean;
813
+ cascade: {
814
+ runs_deleted: number;
815
+ };
816
+ }
817
+ export interface RunResponse {
818
+ success: boolean;
819
+ run: EvalRun;
820
+ }
821
+ export interface RunListResponse {
822
+ success: boolean;
823
+ total: number;
824
+ runs: EvalRun[];
825
+ }
826
+ export interface LiveEventResponse {
827
+ success: boolean;
828
+ event: LiveEvalEvent;
829
+ }
830
+ export interface LiveTargetListResponse {
831
+ success: boolean;
832
+ window: string;
833
+ targets: LiveEvalTarget[];
834
+ }
835
+ export interface LiveEventListResponse {
836
+ success: boolean;
837
+ window: string;
838
+ total: number;
839
+ events: LiveEvalEvent[];
840
+ }
841
+ export interface LiveEventCreateResponse {
842
+ success: boolean;
843
+ event: LiveEvalEvent;
844
+ /** The alerts this event's arrival raised, if any (B2-13). */
845
+ alerts: LiveEvalAlert[];
846
+ }
847
+ export interface LiveAlertListResponse {
848
+ success: boolean;
849
+ window: string;
850
+ alerts: LiveEvalAlert[];
851
+ }
852
+ export interface LiveTargetDeleteResponse {
853
+ success: boolean;
854
+ target_id: string;
855
+ target_type: string;
856
+ deleted_events: number;
857
+ }
858
+ export interface SuccessResponse {
859
+ success: boolean;
860
+ }
861
+ export interface CreateLaunchRequest {
862
+ experiment_ids: string[];
863
+ run_mode?: EvalKind;
864
+ config?: Record<string, unknown>;
865
+ /** The launch this one runs again, at the benchmark version it ran (B5-06). */
866
+ reproduces_launch_id?: string;
867
+ }
868
+ export interface ReviewCaseRequest {
869
+ status: 'passed' | 'failed';
870
+ explanation?: string;
871
+ failure_mode?: string;
872
+ }
873
+ export interface LaunchResponse {
874
+ success: boolean;
875
+ launch: EvalLaunch;
876
+ runs: EvalRun[];
877
+ /** Whether anything executes the queued runs yet. */
878
+ executes?: boolean;
879
+ }
880
+ export interface LaunchListResponse {
881
+ success: boolean;
882
+ total: number;
883
+ launches: EvalLaunch[];
884
+ }
885
+ export interface LaunchCancelResponse {
886
+ success: boolean;
887
+ launch: EvalLaunch;
888
+ cancelled_runs: number;
889
+ }
890
+ export interface RunCancelResponse {
891
+ success: boolean;
892
+ run: EvalRun;
893
+ execution_stopped: boolean;
894
+ }
895
+ export interface CaseResultListResponse {
896
+ success: boolean;
897
+ total: number;
898
+ cases: EvalTaskResult[];
899
+ }
900
+ /** What a message of a run's network says: never its payload. */
901
+ export interface EvalNetworkSummary {
902
+ kind: 'text' | 'object' | 'list' | 'empty';
903
+ /** One scrubbed line of text. */
904
+ preview: string;
905
+ /** An object's field names, never their values. */
906
+ fields: string[];
907
+ items?: number;
908
+ bytes: number;
909
+ /** What was taken out, in words. */
910
+ redactions: string[];
911
+ }
912
+ export interface EvalNetworkMessage {
913
+ id: string;
914
+ from: string;
915
+ to: string;
916
+ type: 'delegation' | 'request' | 'result' | 'review' | 'evaluation';
917
+ taskId: string;
918
+ /** Milliseconds from the run's start. */
919
+ atMs: number;
920
+ durationMs: number;
921
+ status: 'delivered' | 'in-transit' | 'queued' | 'failed';
922
+ summary: EvalNetworkSummary;
923
+ toolCalls: Array<{
924
+ name: string;
925
+ }>;
926
+ links: Record<string, string>;
927
+ correlation: {
928
+ traceId?: string;
929
+ executionId?: string;
930
+ parentMessageId?: string;
931
+ protocolTaskId?: string;
932
+ };
933
+ }
934
+ export interface EvalNetworkTask {
935
+ id: string;
936
+ title: string;
937
+ category: string;
938
+ status: 'passed' | 'review' | 'running' | 'failed' | 'queued';
939
+ agentId: string;
940
+ startedMs: number;
941
+ endedMs: number;
942
+ tokens: number;
943
+ costUsd: number;
944
+ /** What the platform charged, which is credits and not dollars. */
945
+ costCredits: number;
946
+ failureMode?: string;
947
+ finding?: string;
948
+ links: Record<string, string>;
949
+ }
950
+ export interface EvalNetworkAgent {
951
+ id: string;
952
+ role: 'runner' | 'analyst' | 'reviewer' | 'evaluator' | 'agent';
953
+ name: string;
954
+ detail?: string;
955
+ protocol?: 'a2a' | 'acp' | 'datalayer';
956
+ }
957
+ /**
958
+ * A run as the network of agents that worked on it: who asked whom, about
959
+ * which task, when — with summaries and never payloads. Aggregated to a
960
+ * `budget` by the service, which says what it left out.
961
+ */
962
+ export interface EvalRunNetwork {
963
+ run: {
964
+ id: string;
965
+ label: string;
966
+ benchmark: string;
967
+ demo: boolean;
968
+ status: 'queued' | 'running' | 'completed' | 'failed' | 'cancelled';
969
+ counts: Record<EvalNetworkTask['status'], number>;
970
+ startedAt: string;
971
+ durationMs: number;
972
+ tokens: number;
973
+ costUsd: number;
974
+ costCredits: number;
975
+ };
976
+ agents: EvalNetworkAgent[];
977
+ tasks: EvalNetworkTask[];
978
+ messages: EvalNetworkMessage[];
979
+ subject: {
980
+ kind: string;
981
+ ref: string;
982
+ };
983
+ /** A team's tasks that name no execution: stored before the field existed. */
984
+ untraced: number;
985
+ budget: number;
986
+ /** Messages of passed tasks left out to stay within the budget. */
987
+ messagesOmitted: number;
988
+ /** Tasks past the most a network reads. */
989
+ tasksOmitted: number;
990
+ }
991
+ /** One message of a run's network, in full: more parts, each a summary. */
992
+ export interface RunNetworkMessageResponse {
993
+ success: boolean;
994
+ message: {
995
+ id: string;
996
+ parts: Array<{
997
+ label: string;
998
+ summary: EvalNetworkSummary;
999
+ }>;
1000
+ };
1001
+ }
1002
+ /** A launch's runs, each as one line of its own network. */
1003
+ export interface EvalLaunchNetwork {
1004
+ launchId: string;
1005
+ status: string;
1006
+ counts: Record<EvalNetworkTask['status'], number>;
1007
+ runs: Array<{
1008
+ runId: string;
1009
+ experimentId: string;
1010
+ status: EvalRunNetwork['run']['status'];
1011
+ subject: {
1012
+ kind: string;
1013
+ ref: string;
1014
+ };
1015
+ counts: Record<EvalNetworkTask['status'], number>;
1016
+ tasks: number;
1017
+ startedAt: string;
1018
+ endedAt: string;
1019
+ }>;
1020
+ }
1021
+ export interface LaunchNetworkResponse {
1022
+ success: boolean;
1023
+ network: EvalLaunchNetwork;
1024
+ }
1025
+ export interface RunNetworkResponse {
1026
+ success: boolean;
1027
+ network: EvalRunNetwork;
1028
+ }
1029
+ export interface CaseResultResponse {
1030
+ success: boolean;
1031
+ case: EvalTaskResult;
1032
+ }
1033
+ export interface ImportEvalsetRequest {
1034
+ /** An `*.evalset.json` spec, as the CLI and the action read it. */
1035
+ spec: Record<string, unknown>;
1036
+ run_environment?: EvalRunEnvironment;
1037
+ }
1038
+ export interface ImportEvalsetResponse extends EvalsetResponse {
1039
+ /** Evaluators the platform cannot run, left out of the evalset. */
1040
+ unsupported_evaluators: string[];
1041
+ }
1042
+ export interface EvalsetVersionListResponse {
1043
+ success: boolean;
1044
+ total: number;
1045
+ versions: EvalsetVersion[];
1046
+ }
1047
+ export interface EvalsetVersionResponse {
1048
+ success: boolean;
1049
+ version: EvalsetVersion;
1050
+ }
1051
+ export type InvestigationScope = 'case' | 'run' | 'launch' | 'window';
1052
+ export type InvestigationStatus = 'open' | 'closed';
1053
+ /** The restored sandbox, per section 12.6. */
1054
+ export type SandboxState = 'available' | 'restoring' | 'running' | 'expired' | 'none';
1055
+ /** The work a person does on a benchmark result: one per task, run, launch or window. */
1056
+ export interface EvalInvestigation {
1057
+ id: string;
1058
+ owner_uid: string;
1059
+ account_uid: string;
1060
+ scope: InvestigationScope;
1061
+ evalset_id: string;
1062
+ launch_id: string;
1063
+ run_id: string;
1064
+ case_id: string;
1065
+ window_id: string;
1066
+ title: string;
1067
+ status: InvestigationStatus;
1068
+ decision: string;
1069
+ /** The report document (spacer lexical) the investigation is written in. */
1070
+ document_uid: string;
1071
+ /** The notebook the investigation works in; a copy, never the evidence. */
1072
+ notebook_uid: string;
1073
+ /** The task's evidence notebook, written by the run. */
1074
+ evidence_notebook_uid: string;
1075
+ sandbox_snapshot_uid: string;
1076
+ /** The restored sandbox while it lives. */
1077
+ runtime_name: string;
1078
+ sandbox_workflow_uid: string;
1079
+ sandbox_state: SandboxState;
1080
+ created_by_uid: string;
1081
+ assignees: string[];
1082
+ /** Published to the library by its owner, with what it is written in. */
1083
+ is_public: boolean;
1084
+ metadata: Record<string, unknown>;
1085
+ created_at: string | null;
1086
+ updated_at: string | null;
1087
+ }
1088
+ export interface InvestigationResponse extends SuccessResponse {
1089
+ investigation: EvalInvestigation;
1090
+ }
1091
+ export interface TaskInvestigationResponse extends InvestigationResponse {
1092
+ case: EvalTaskResult;
1093
+ }
1094
+ export interface InvestigationListResponse extends SuccessResponse {
1095
+ total: number;
1096
+ investigations: EvalInvestigation[];
1097
+ }
1098
+ export interface UpdateInvestigationRequest {
1099
+ status?: InvestigationStatus;
1100
+ decision?: string;
1101
+ title?: string;
1102
+ assignees?: string[];
1103
+ }
1104
+ /**
1105
+ * A view of an investigation's page kept under a name for everybody on it
1106
+ * (B4-11), in the investigation's `metadata.views`.
1107
+ */
1108
+ export interface SavedInvestigationView {
1109
+ name: string;
1110
+ /** What the page shows: `surface`, and `block`, `case` or `cell`. */
1111
+ query: string;
1112
+ saved_by_uid?: string;
1113
+ saved_at?: string;
1114
+ }
1115
+ export interface SaveInvestigationViewRequest {
1116
+ name: string;
1117
+ query: string;
1118
+ }
1119
+ export interface ResumeSandboxRequest {
1120
+ /** Minutes the restored sandbox is reserved for. */
1121
+ time_reservation?: number;
1122
+ environment?: string;
1123
+ }
1124
+ export interface ResumeSandboxResponse extends InvestigationResponse {
1125
+ sandbox: {
1126
+ state: SandboxState;
1127
+ runtime_name: string;
1128
+ workflow_uid?: string;
1129
+ reused?: boolean;
1130
+ };
1131
+ }
1132
+ /** A serialized Lexical editor state, as the editor writes it. */
1133
+ export interface LexicalReportResponse extends SuccessResponse {
1134
+ document: {
1135
+ root: Record<string, unknown>;
1136
+ };
1137
+ }
1138
+ export interface ReportDocumentResponse extends SuccessResponse {
1139
+ document_uid: string;
1140
+ launch: EvalLaunch;
1141
+ }
1142
+ /** The records that carry grants, as their routes name them. */
1143
+ export type SharedEvalsRecord = 'evalsets' | 'launches' | 'investigations' | 'reports';
1144
+ /**
1145
+ * The levels a grant gives, lowest first, each allowing what the ones before
1146
+ * it allow: Viewer, Reviewer, Runner, Editor.
1147
+ */
1148
+ export type EvalsAccessLevel = 'view' | 'review' | 'execute' | 'update';
1149
+ /** A level, or the owner — which is never granted: only an owner shares or publishes. */
1150
+ export type EvalsRole = EvalsAccessLevel | 'owner';
1151
+ export interface EvalsPrincipals {
1152
+ userUids: string[];
1153
+ teamUids: string[];
1154
+ organizationUids: string[];
1155
+ }
1156
+ export type EvalsSharingUpdate = Partial<Record<EvalsAccessLevel, Partial<EvalsPrincipals>>>;
1157
+ export interface EvalsSharingResponse extends SuccessResponse {
1158
+ sharing: {
1159
+ kind: 'evalset' | 'launch' | 'investigation';
1160
+ uid: string;
1161
+ /** The account the record belongs to. */
1162
+ owner_uid: string;
1163
+ access: Record<EvalsAccessLevel, EvalsPrincipals>;
1164
+ shared: boolean;
1165
+ };
1166
+ }
1167
+ /**
1168
+ * How far a record reaches (BENCHMARKS.md, B5-09, section 14.5), narrowest
1169
+ * first. Derived from the grants and the public flag rather than stored, so it
1170
+ * cannot disagree with who may actually read the thing.
1171
+ */
1172
+ export type EvalsVisibility = 'private' | 'users' | 'team' | 'organization' | 'public';
1173
+ export interface EvalsPermissionsResponse extends SuccessResponse {
1174
+ kind: string;
1175
+ uid: string;
1176
+ role: EvalsRole;
1177
+ permissions: Record<EvalsRole, boolean>;
1178
+ /** The widest thing that is true of this record. */
1179
+ visibility: EvalsVisibility;
1180
+ }
1181
+ /** The states of a report (section 13.3), in the order a report goes through them. */
1182
+ export type ReportState = 'draft' | 'in_review' | 'approved' | 'published_private' | 'published_public' | 'superseded';
1183
+ /** A report over runs of one benchmark: its document and its state. */
1184
+ export interface EvalReport {
1185
+ id: string;
1186
+ owner_uid: string;
1187
+ account_uid: string;
1188
+ evalset_id: string;
1189
+ evalset_version: number | null;
1190
+ launch_ids: string[];
1191
+ run_ids: string[];
1192
+ /** The Lexical document in the benchmarks space people read and edit. */
1193
+ document_uid: string;
1194
+ title: string;
1195
+ state: ReportState;
1196
+ version: number;
1197
+ /** Who was asked to approve it. */
1198
+ reviewer_uids: string[];
1199
+ approved_by_uid: string;
1200
+ approved_at: string | null;
1201
+ /** The version of the document kept when it was approved. */
1202
+ approved_version_uid: string;
1203
+ published_at: string | null;
1204
+ supersedes_uid: string;
1205
+ superseded_by_uid: string;
1206
+ created_by_uid: string;
1207
+ created_at: string | null;
1208
+ updated_at: string | null;
1209
+ }
1210
+ export interface ReportResponse extends SuccessResponse {
1211
+ report: EvalReport;
1212
+ }
1213
+ export interface ReportListResponse extends SuccessResponse {
1214
+ total: number;
1215
+ reports: EvalReport[];
1216
+ }
1217
+ export interface CreateReportRequest {
1218
+ evalset_id: string;
1219
+ launch_ids: string[];
1220
+ title?: string;
1221
+ }
1222
+ /** A move of a report; `superseded` is not one, a regeneration is. */
1223
+ export interface MoveReportRequest {
1224
+ state: Exclude<ReportState, 'superseded'>;
1225
+ /** Who is asked to approve it, when it is sent for review. */
1226
+ reviewer_uids?: string[];
1227
+ /** What the approval says. */
1228
+ message?: string;
1229
+ }
1230
+ export interface RegenerateReportResponse extends ReportResponse {
1231
+ superseded: EvalReport;
1232
+ }
1233
+ export type ReportsQuery = {
1234
+ evalset_id?: string;
1235
+ launch_id?: string;
1236
+ state?: ReportState;
1237
+ limit?: number;
1238
+ offset?: number;
1239
+ };
1240
+ export type DecisionKind = 'accepted_regression' | 'expected_change' | 'evaluator_issue' | 'data_issue' | 'action_required';
1241
+ export type DecisionOutcome = 'approved' | 'blocked' | 'accepted_with_limitations';
1242
+ /** What a decision is about: a block of a report, a task, a run, a launch. */
1243
+ export type DecisionScope = 'block' | 'case' | 'run' | 'launch';
1244
+ /** The records a decision is made on, as their routes name them. */
1245
+ export type DecisionSubject = 'reports' | 'investigations';
1246
+ /** A decision as it was made; decisions are appended, never edited. */
1247
+ export interface EvalDecision {
1248
+ id: string;
1249
+ owner_uid: string;
1250
+ account_uid: string;
1251
+ subject: 'report' | 'investigation';
1252
+ subject_uid: string;
1253
+ kind: DecisionKind;
1254
+ outcome: DecisionOutcome;
1255
+ scope: DecisionScope;
1256
+ /** What in the scope: the block, the task, the run or the launch. */
1257
+ scope_ref: string;
1258
+ note: string;
1259
+ decided_by_uid: string;
1260
+ decided_at: string | null;
1261
+ /** The comment thread the decision resolves. */
1262
+ thread_uid: string;
1263
+ evalset_id: string;
1264
+ launch_ids: string[];
1265
+ }
1266
+ export interface DecisionRequest {
1267
+ kind: DecisionKind;
1268
+ outcome: DecisionOutcome;
1269
+ scope: DecisionScope;
1270
+ scope_ref?: string;
1271
+ note?: string;
1272
+ thread_uid?: string;
1273
+ }
1274
+ export interface DecisionResponse extends SuccessResponse {
1275
+ decision: EvalDecision;
1276
+ }
1277
+ export interface DecisionListResponse extends SuccessResponse {
1278
+ total: number;
1279
+ decisions: EvalDecision[];
1280
+ }
1281
+ /** A report or an investigation somebody shared with the caller. */
1282
+ export interface SharedEvalsItem {
1283
+ kind: 'report' | 'investigation';
1284
+ uid: string;
1285
+ title: string;
1286
+ owner_uid: string;
1287
+ /** The level the grants give the caller. */
1288
+ role: EvalsAccessLevel;
1289
+ /** The report's state or the investigation's status. */
1290
+ state: string;
1291
+ /** The page of the app it opens. */
1292
+ link: string;
1293
+ updated_at: string | null;
1294
+ }
1295
+ export interface SharedWithMeResponse extends SuccessResponse {
1296
+ total: number;
1297
+ shared: SharedEvalsItem[];
1298
+ }
1299
+ /** A report file CI produced, kept on its benchmark as it was. */
1300
+ export interface EvalReportImport {
1301
+ id: string;
1302
+ owner_uid: string;
1303
+ account_uid: string;
1304
+ evalset_id: string;
1305
+ format: 'csv' | 'markdown';
1306
+ name: string;
1307
+ run_ids: string[];
1308
+ launch_ids: string[];
1309
+ /** Run ids the file names that are no run of this benchmark here. */
1310
+ unmatched_run_ids: string[];
1311
+ imported_by_uid: string;
1312
+ /** The live report continued from it, once one is. */
1313
+ live_report_uid: string;
1314
+ investigation_uid: string;
1315
+ /** The file's text; left out of a listing. */
1316
+ content: string | null;
1317
+ created_at: string | null;
1318
+ updated_at: string | null;
1319
+ }
1320
+ export interface ReportImportRequest {
1321
+ format: 'csv' | 'markdown';
1322
+ content: string;
1323
+ name?: string;
1324
+ }
1325
+ export interface ReportImportResponse extends SuccessResponse {
1326
+ import: EvalReportImport;
1327
+ }
1328
+ export interface ReportImportListResponse extends SuccessResponse {
1329
+ total: number;
1330
+ imports: EvalReportImport[];
1331
+ }
1332
+ export interface ContinueReportImportResponse extends ReportImportResponse {
1333
+ report_link: string;
1334
+ investigation_link: string;
1335
+ }