@evolvingmachines/sdk 0.0.51 → 0.0.52-launch-round-1.20260803.37a56fe

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -0,0 +1,1582 @@
1
+ /**
2
+ * Public types for the hosted evaluations API — datasets, jobs, trials, agents.
3
+ *
4
+ * THE VOCABULARY IS THE WIRE'S. Every field on a wire-shaped object below is
5
+ * spelled exactly as spec/openapi.yaml spells it (snake_case), so the spec
6
+ * reads as the SDK's own field reference and nothing is ever lost in a casing
7
+ * translation. The only camelCase keys are the four frozen historical spots
8
+ * the spec names: the page envelope (`items`/`nextCursor`/`hasMore`), the job
9
+ * body's `trials.byStatus`, the compare response's `taskMatrix`, and the error
10
+ * envelope (`retryAfterSec`/`requestId`). That freeze is a WIRE law, not a
11
+ * property-name law: both SDKs send and receive those keys camelCase, this SDK
12
+ * also exposes them verbatim, and the Python SDK maps them to snake_case
13
+ * attributes (`next_cursor`/`has_more`/`by_status`/`task_matrix`,
14
+ * `retry_after_sec`/`request_id`). SDK-side controls that never touch the
15
+ * wire — client config, delivery options, callbacks — stay
16
+ * TypeScript-idiomatic camelCase.
17
+ */
18
+ /** Configuration for the datasets() / agents() / jobs() / trials() factories */
19
+ interface HostedClientConfig {
20
+ /** API key (default: process.env.EVOLVE_API_KEY) */
21
+ apiKey?: string;
22
+ /** API base URL override (default: the Evolve dashboard API) */
23
+ baseUrl?: string;
24
+ }
25
+ /**
26
+ * ONE page shape for every collection on this surface — top level or nested.
27
+ *
28
+ * `nextCursor` means one thing everywhere: pass it back as the next call's
29
+ * `cursor` for the next page, and `null` means there is no next page. It never
30
+ * echoes where you already are, so a poller can always tell it has caught up.
31
+ * The three envelope keys are frozen verbatim on the wire.
32
+ */
33
+ interface Page<T> {
34
+ items: T[];
35
+ nextCursor: string | null;
36
+ hasMore: boolean;
37
+ }
38
+ /**
39
+ * A value you can `await`, with the rest of the promise surface attached.
40
+ *
41
+ * The dual-use handles below were `PromiseLike` alone, which is enough for
42
+ * `await` and nothing else — so `client.list().catch(...)` was a compile error
43
+ * two lines after `await client.list()` compiled fine, and `.finally()` for a
44
+ * spinner was unavailable. A handle that is 90% of a promise is worse than one
45
+ * that is none of it, because the missing 10% is only discovered at the call
46
+ * site that needed it.
47
+ *
48
+ * `then`/`catch`/`finally` all return real Promises, so anything chained off a
49
+ * handle behaves exactly like promise code from that point on.
50
+ */
51
+ interface Awaitable<T> extends PromiseLike<T> {
52
+ catch<TResult = never>(onrejected?: ((reason: unknown) => TResult | PromiseLike<TResult>) | null): Promise<T | TResult>;
53
+ finally(onfinally?: (() => void) | null): Promise<T>;
54
+ }
55
+ /** Cursor + page-size options, accepted by every paged call */
56
+ interface PageOptions {
57
+ /** Max items per page */
58
+ limit?: number;
59
+ /** Cursor from a previous page's nextCursor */
60
+ cursor?: string;
61
+ }
62
+ /**
63
+ * Job lifecycle status (wire values, as the API emits them).
64
+ * Terminal: COMPLETED, CANCELLED, FAILED.
65
+ */
66
+ type JobStatus = "QUEUED" | "RUNNING" | "CANCELLING" | "COMPLETED" | "CANCELLED" | "FAILED";
67
+ /**
68
+ * Trial status law: a valid reward (including 0) = SCORED; verifier crash or
69
+ * out-of-domain reward = SCORING_ERROR (never a fabricated zero);
70
+ * INFRASTRUCTURE_ERROR: the trial was lost before a result was recorded;
71
+ * INDETERMINATE: the platform cannot tell whether the trial completed.
72
+ *
73
+ * A runtime value (not only a type), like TRIAL_ARTIFACT_STREAMS, so the CLI
74
+ * can validate a `--status` filter against this list instead of a second copy.
75
+ */
76
+ declare const TRIAL_STATUSES: readonly ["QUEUED", "RUNNING", "SCORING", "SCORED", "SCORING_ERROR", "INFRASTRUCTURE_ERROR", "INDETERMINATE", "CANCELLED"];
77
+ /** One trial lifecycle status — see TRIAL_STATUSES for the law. */
78
+ type TrialStatus = (typeof TRIAL_STATUSES)[number];
79
+ /**
80
+ * Sandbox provider a hosted job runs on. Named `EvalSandboxProvider` to avoid
81
+ * colliding with the core SDK's `SandboxProvider` (the sandbox-abstraction
82
+ * interface). A runtime value for the same reason as TRIAL_STATUSES: the CLI
83
+ * validates `-e/--env` against it.
84
+ */
85
+ declare const EVAL_SANDBOX_PROVIDERS: readonly ["e2b", "daytona", "modal"];
86
+ /** One sandbox provider — see EVAL_SANDBOX_PROVIDERS. */
87
+ type EvalSandboxProvider = (typeof EVAL_SANDBOX_PROVIDERS)[number];
88
+ /**
89
+ * Which lane a settled trial's `agent_result.cost_usd` came from. Only
90
+ * `"measured"` is final. `"measured_provisional"` is a real gateway reading
91
+ * taken inside its asynchronous spend flush — an honest floor a deferred pass
92
+ * later confirms or raises into `"measured"`. `"assumed_cap"` means nobody
93
+ * measured this trial: the figure it carries is zero, a placeholder and never
94
+ * the cap (the platform under-bills rather than publish an invented number),
95
+ * replaced when a real reading lands. Read anything but `"measured"` as not
96
+ * yet final.
97
+ */
98
+ type SpendSource = "measured" | "measured_provisional" | "assumed_cap";
99
+ /** Where a trial's verifier executed: inside the agent's environment, or a separate one */
100
+ type VerifierEnvironmentMode = "shared" | "separate";
101
+ /**
102
+ * Which step a RUNNING trial is in, so a polling caller can tell a slow build
103
+ * from a slow agent — RUNNING alone cannot.
104
+ */
105
+ type AttemptPhase = "prepare" | "build" | "boot" | "install" | "agent" | "verify" | "persist";
106
+ /**
107
+ * Trial count histogram by status. EVERY status is present, zeros included, so
108
+ * a status bar can be drawn straight off the response without hardcoding the
109
+ * enum and discovering a new status only when a bar goes missing.
110
+ */
111
+ type TrialCounts = Record<TrialStatus, number>;
112
+ /**
113
+ * The one "how many" shape: a total plus the zeros-included histogram.
114
+ * `byStatus` is one of the four frozen camelCase wire keys.
115
+ */
116
+ interface TrialStatusTally {
117
+ total: number;
118
+ byStatus: TrialCounts;
119
+ }
120
+ /** A resolved dataset reference as echoed on job bodies. */
121
+ interface DatasetRef {
122
+ name: string;
123
+ version: string;
124
+ }
125
+ /**
126
+ * One dataset a job runs, with per-dataset task filters. `task_names` and
127
+ * `exclude_task_names` are glob patterns; `n_tasks` caps the task count AFTER
128
+ * filtering. A bare `name` resolves to the active version (`no_active_version`
129
+ * when none).
130
+ */
131
+ interface DatasetSelector {
132
+ /** Catalog dataset name. */
133
+ name: string;
134
+ /** Pin a version; omitted, the active version is used. */
135
+ version?: string;
136
+ /** Include filter — glob patterns over task names. */
137
+ task_names?: string[];
138
+ /** Exclude filter — glob patterns over task names. */
139
+ exclude_task_names?: string[];
140
+ /** Cap the task count after filters are applied. */
141
+ n_tasks?: number;
142
+ }
143
+ /**
144
+ * One agent arm of a job: an agent (built-in or registered) plus a model. A
145
+ * model is always required; the server applies no default.
146
+ *
147
+ * `version` pins an agent version; omitted, the platform resolves the latest
148
+ * supported (`agent_version_not_found` when a pin cannot resolve). The version
149
+ * that actually RAN is recorded on every trial as `agent_info.version`.
150
+ *
151
+ * `reasoning_effort` is the platform extension: declared effort, PART OF THE
152
+ * ARM'S IDENTITY like the agent, the model and the version pin — the same
153
+ * agent and model at "low" and at "high" are two systems, and they
154
+ * de-duplicate separately. Accepted values are published by /api/meta; an
155
+ * effort the agent cannot apply is refused at creation, never recorded and
156
+ * silently dropped.
157
+ */
158
+ interface AgentArmInput {
159
+ /** Agent name — a built-in or one registered under /api/agents. */
160
+ name: string;
161
+ model_name: string;
162
+ version?: string | null;
163
+ reasoning_effort?: string | null;
164
+ }
165
+ /** One agent arm as echoed on job bodies (requested pin; null = took the latest). */
166
+ interface AgentArm {
167
+ name: string;
168
+ model_name: string;
169
+ version: string | null;
170
+ reasoning_effort: string | null;
171
+ }
172
+ /**
173
+ * Provenance of a derived job. `action: "regrade"` = verifier-only re-run of
174
+ * the source; `action: "resume"` (platform extension) = new job over the
175
+ * source's failed and stopped trials. `type` is always "hub" on this hosted
176
+ * surface.
177
+ */
178
+ interface SourceJob {
179
+ action: "regrade" | "resume";
180
+ type: "hub";
181
+ job_id: string;
182
+ }
183
+ /** The job-creation body — POST /api/jobs. */
184
+ interface JobCreate {
185
+ /** User-facing label; server-generated when omitted. */
186
+ job_name?: string;
187
+ datasets: DatasetSelector[];
188
+ agents: AgentArmInput[];
189
+ /** Attempts per task per agent arm (default 1, max 100). */
190
+ n_attempts?: number;
191
+ /** Parallel trials across the job (default 4, max 16). */
192
+ n_concurrent_trials?: number;
193
+ /**
194
+ * Per-trial spend cap in USD, minted onto each trial's gateway key — the
195
+ * platform's ONLY spend enforcement (there is no job-wide budget). Omitted,
196
+ * the server applies its published default ($200 unless the operator tuned
197
+ * it); the response echoes the RESOLVED cap either way, and states the
198
+ * resulting worst case for the job as a whole.
199
+ */
200
+ max_trial_spend_usd?: number;
201
+ /** Sandbox provider to run on (optional; server default: `e2b`). */
202
+ sandbox_provider?: EvalSandboxProvider;
203
+ /**
204
+ * Env injected into every agent run — a pass-through slot: the client sends
205
+ * it verbatim and the server owns acceptance (refused where unsupported,
206
+ * never silently dropped).
207
+ */
208
+ agent_env?: Record<string, string>;
209
+ /** Env injected into every verifier run — same pass-through contract. */
210
+ verifier_env?: Record<string, string>;
211
+ }
212
+ /** Body of POST /api/jobs/{jobId}/resume. */
213
+ interface ResumeRequest {
214
+ /**
215
+ * Which failures to resume, matched against
216
+ * `exception_info.exception_type`. Omitted, the default set is
217
+ * ["ScoringError", "InfrastructureError", "IncompleteTrialError"] plus
218
+ * stopped trials (settled CANCELLED, exception type "CancelledError")
219
+ * and still-QUEUED trials of a cancelled source.
220
+ */
221
+ filter_error_types?: string[];
222
+ }
223
+ /**
224
+ * Optional filter narrowing which trials a job-level regrade re-runs.
225
+ * Omitted, every regradable trial is regraded.
226
+ */
227
+ interface RegradeRequest {
228
+ statuses?: TrialStatus[];
229
+ /** Restrict to one task's trials. */
230
+ task_name?: string;
231
+ }
232
+ /**
233
+ * Per-(agent, model, dataset) statistics. The evals key format is
234
+ * `{agent}__{model}__{dataset}` — the dataset ref is always the LAST `__`
235
+ * segment, which is where Harbor-compatible readers recover it — with the
236
+ * platform extension of an `__{effort}` segment inserted BEFORE the dataset
237
+ * when a declared reasoning effort is part of the arm identity:
238
+ * `{agent}__{model}__{effort}__{dataset}`.
239
+ */
240
+ interface AgentDatasetStats {
241
+ /** Trials that produced a rewards map — rewarded, not merely settled. */
242
+ n_trials?: number;
243
+ /** Trials carrying `exception_info` — indeterminate and cancelled included. */
244
+ n_errors?: number;
245
+ /**
246
+ * Metric results (a mean entry per arm today: the primary reward averaged
247
+ * over EVERY trial of the group, unrewarded trials counting 0); open objects.
248
+ */
249
+ metrics?: Record<string, unknown>[];
250
+ /**
251
+ * pass@k slot — keys are k as strings, values in [0,1]. Present and empty
252
+ * until the platform computes it; the slot exists so adding the statistic is
253
+ * not a wire change.
254
+ */
255
+ pass_at_k?: Record<string, number>;
256
+ /** reward key -> reward value -> trial identifiers. */
257
+ reward_stats?: Record<string, Record<string, string[]>>;
258
+ /** exception type -> trial identifiers. */
259
+ exception_stats?: Record<string, string[]>;
260
+ }
261
+ /**
262
+ * Aggregate statistics of a job. Progress counters, token totals, and measured
263
+ * cost. The `n_*` counters are CUMULATIVE, Harbor-style: errored trials are a
264
+ * subset of completed, cancelled a subset of errored — a cancelled trial
265
+ * counts in all three. The disjoint per-status breakdown rides
266
+ * `Job.trials.byStatus`. `cost_usd` is what the trials actually spent so far —
267
+ * reporting, never a gate (enforcement is the per-trial cap).
268
+ */
269
+ interface JobStats {
270
+ /** Cumulative: every trial that produced a result — errored and cancelled included. */
271
+ n_completed_trials?: number;
272
+ /** Cumulative: every completed trial carrying `exception_info`, cancelled included. */
273
+ n_errored_trials?: number;
274
+ n_running_trials?: number;
275
+ n_pending_trials?: number;
276
+ /** A subset of `n_errored_trials`. */
277
+ n_cancelled_trials?: number;
278
+ n_retries?: number;
279
+ /** Keyed `{agent}__{model}__{dataset}` — dataset ref last, optional effort segment before it. */
280
+ evals?: Record<string, AgentDatasetStats>;
281
+ /** Total input tokens (cache included); null until recorded. */
282
+ n_input_tokens?: number | null;
283
+ n_cache_tokens?: number | null;
284
+ n_output_tokens?: number | null;
285
+ /** Measured spend across settled trials; null before any settled. */
286
+ cost_usd?: number | null;
287
+ }
288
+ /**
289
+ * Why a job FAILED — deliberately NOT under the key `error`, which on this
290
+ * surface always means "this request failed". `if (body.error) throw` stays
291
+ * correct on a healthy 200 read of a failed job.
292
+ */
293
+ interface JobFailure {
294
+ /** `job_execution_failed` when the runner recorded no code. */
295
+ code: string;
296
+ message: string;
297
+ }
298
+ /**
299
+ * THE job body — the same shape from create, get, list items, cancel, resume,
300
+ * and regrade responses; no field appears on some responses and not others.
301
+ */
302
+ interface Job {
303
+ id: string;
304
+ /** User-facing label. */
305
+ job_name: string;
306
+ status: JobStatus;
307
+ /** The resolved dataset references this job ran. */
308
+ datasets: DatasetRef[];
309
+ agents: AgentArm[];
310
+ n_attempts: number;
311
+ n_concurrent_trials: number;
312
+ /** The resolved per-trial cap every trial key was minted with. */
313
+ max_trial_spend_usd: number;
314
+ /**
315
+ * The most this job can cost: every trial spending its whole cap. Stated
316
+ * outright — the per-trial cap is the only enforcement, so the product is
317
+ * the number someone approving a 500-trial run actually needs to see.
318
+ */
319
+ worst_case_spend_usd: number;
320
+ sandbox_provider: EvalSandboxProvider;
321
+ /** Entity cardinality only — things with no status of their own. */
322
+ counts: {
323
+ agents: number;
324
+ tasks: number;
325
+ };
326
+ n_total_trials: number;
327
+ /** The zeros-included 8-status histogram, beside the coarser counters in `stats`. */
328
+ trials: TrialStatusTally;
329
+ stats: JobStats;
330
+ /** Why the job FAILED, or null. Never the key `error` — see JobFailure. */
331
+ failure: JobFailure | null;
332
+ /** Empty for an original job. */
333
+ source_jobs: SourceJob[];
334
+ /** Derived: any source_jobs entry with action "regrade". */
335
+ is_regrade: boolean;
336
+ /** True only on a response that replayed an existing job for an Idempotency-Key. */
337
+ idempotent_replay: boolean;
338
+ started_at: string;
339
+ updated_at: string;
340
+ /** Null while the job is live. */
341
+ finished_at: string | null;
342
+ }
343
+ /**
344
+ * A phase's wall-clock as a start/stop pair (never a duration). Either bound
345
+ * is null while the phase has not reached it.
346
+ */
347
+ interface TimingInfo {
348
+ started_at: string | null;
349
+ finished_at: string | null;
350
+ }
351
+ interface ModelInfo {
352
+ name: string;
353
+ /** Null means "not specified", never "unknown provider". */
354
+ provider?: string | null;
355
+ }
356
+ /**
357
+ * The agent that ran a trial. `version` is the version actually RESOLVED and
358
+ * used (null until resolved) — the requested pin lives on the job's
359
+ * `agents[].version`. `reasoning_effort` is the platform's arm-identity
360
+ * extension.
361
+ */
362
+ interface AgentInfo {
363
+ name: string;
364
+ version: string | null;
365
+ model_info: ModelInfo;
366
+ reasoning_effort?: string | null;
367
+ }
368
+ /**
369
+ * What the agent phase produced and consumed. `n_input_tokens` includes cache
370
+ * tokens. `cost_usd` is the settled spend (see `spend_source` on the trial for
371
+ * which lane it came from, and whether it is final). `metadata` carries open
372
+ * per-run detail:
373
+ * the harness bundle digest and runtime, the network mode the trial ran under
374
+ * and where that decision came from, and any harness-reported usage detail.
375
+ */
376
+ interface AgentResult {
377
+ n_input_tokens?: number | null;
378
+ n_cache_tokens?: number | null;
379
+ n_output_tokens?: number | null;
380
+ /** Null until the trial has executed; null never means $0. */
381
+ cost_usd?: number | null;
382
+ /** Reserved for token-level rollout detail; null today. */
383
+ rollout_details?: Record<string, unknown>[] | null;
384
+ metadata?: Record<string, unknown> | null;
385
+ }
386
+ /**
387
+ * The verifier's rewards map. The primary-reward convention: the value under
388
+ * the key "reward"; else, when exactly one key exists, that value; else no
389
+ * primary reward. Zero is a reward.
390
+ */
391
+ interface VerifierResult {
392
+ rewards?: Record<string, number> | null;
393
+ }
394
+ /**
395
+ * Why a trial failed, when it did. `exception_type` is one of the platform's
396
+ * stable failure names (ScoringError, InfrastructureError, CancelledError,
397
+ * IncompleteTrialError) — but filter with `Trial.status`, which is the primary
398
+ * key for failure classes; this is the detail.
399
+ */
400
+ interface ExceptionInfo {
401
+ exception_type: string;
402
+ /** Truncated to 2000 chars on list rows; full on the detail route. */
403
+ exception_message: string;
404
+ /** Empty when the platform recorded no traceback. */
405
+ exception_traceback?: string;
406
+ occurred_at: string;
407
+ }
408
+ /**
409
+ * Placeholder for multi-step tasks. Always null on trials today — declared so
410
+ * multi-step lands without a wire change.
411
+ */
412
+ interface StepResult {
413
+ step_name?: string;
414
+ agent_result?: AgentResult | null;
415
+ verifier_result?: VerifierResult | null;
416
+ exception_info?: ExceptionInfo | null;
417
+ agent_execution?: TimingInfo | null;
418
+ verifier?: TimingInfo | null;
419
+ }
420
+ /**
421
+ * The ONE public trial shape, shared verbatim by list rows and the detail
422
+ * route (detail returns `exception_info.exception_message` untruncated — the
423
+ * only documented difference). A trial id is globally addressable; `job_id` is
424
+ * the reverse pointer.
425
+ *
426
+ * Execution facts (`sandbox_provider`, `verifier_environment_mode`,
427
+ * `agent_result.cost_usd`, `spend_source`) are null until the trial has
428
+ * actually executed: a QUEUED or CANCELLED trial never ran, so null means
429
+ * "did not run" and never zero.
430
+ */
431
+ interface Trial {
432
+ id: string;
433
+ job_id: string;
434
+ task_name: string;
435
+ /** The dataset this trial's task came from. */
436
+ source: string;
437
+ agent_info: AgentInfo;
438
+ /** Attempt index within the arm (1..n_attempts). */
439
+ attempt: number;
440
+ status: TrialStatus;
441
+ /**
442
+ * Convenience primary reward derived from `verifier_result.rewards` by the
443
+ * primary-reward convention. Zero is a reward; null means the trial did not
444
+ * score.
445
+ */
446
+ reward: number | null;
447
+ verifier_result: VerifierResult | null;
448
+ exception_info: ExceptionInfo | null;
449
+ agent_result: AgentResult | null;
450
+ environment_setup: TimingInfo | null;
451
+ agent_setup: TimingInfo | null;
452
+ agent_execution: TimingInfo | null;
453
+ verifier: TimingInfo | null;
454
+ /** Multi-step placeholder; null today. */
455
+ step_results: StepResult[] | null;
456
+ /** Which lane `agent_result.cost_usd` came from — see SpendSource; only "measured" is final. */
457
+ spend_source: SpendSource | null;
458
+ /**
459
+ * A mid-run LOWER BOUND on spend, never the trial's cost. Only ever climbs
460
+ * while the trial runs, and is CLEARED when the trial settles — on a
461
+ * terminal trial read `agent_result.cost_usd` and `spend_source`; those are
462
+ * the settled truth. Null is "no reading yet", never $0.
463
+ */
464
+ live_spent_usd: number | null;
465
+ /** When that reading was taken — show its age, never the figure alone. */
466
+ live_spend_at: string | null;
467
+ /**
468
+ * The cap THIS trial's gateway key carried — history, which can differ from
469
+ * the job's current cap for rows settled before a change.
470
+ */
471
+ max_trial_spend_usd: number | null;
472
+ sandbox_provider: EvalSandboxProvider | null;
473
+ /** Provider id of the box the agent executed in; null when none booted. */
474
+ sandbox_id: string | null;
475
+ /** The separate verifier box; null in shared mode or when never reached. */
476
+ verifier_sandbox_id: string | null;
477
+ verifier_environment_mode: VerifierEnvironmentMode | null;
478
+ /**
479
+ * Which step a RUNNING trial is in, so a polling caller can tell a slow
480
+ * build from a slow agent. Null when the trial is not mid-phase.
481
+ */
482
+ attempt_phase: AttemptPhase | null;
483
+ /** Reference to the agent session/trace, when recorded. */
484
+ session_ref: string | null;
485
+ started_at: string | null;
486
+ finished_at: string | null;
487
+ }
488
+ /** Per-trial outcome of POST /api/trials/stop; every requested id appears in exactly one list. */
489
+ interface StopResponse {
490
+ /** Trials killed and settled by this request, with their settled rows. */
491
+ stopped: Trial[];
492
+ /** Ids that were already terminal; untouched. */
493
+ already_terminal: string[];
494
+ /** Ids that do not exist or are not the caller's. */
495
+ not_found: string[];
496
+ }
497
+ /**
498
+ * One parsed trace event of a trial's transcript. `seq` orders the stream and
499
+ * is the paging cursor. `data` is the harness-native payload, deliberately
500
+ * open.
501
+ */
502
+ interface TraceEvent {
503
+ /** Monotonic sequence number (the resume position). */
504
+ seq: number;
505
+ type: string;
506
+ data: Record<string, unknown>;
507
+ }
508
+ /**
509
+ * One page of a trial's trace — trials().trace().
510
+ *
511
+ * Same envelope as every other collection, and nextCursor means the same
512
+ * thing: pass it back as { cursor } for the next page, and NULL MEANS CAUGHT
513
+ * UP. To resume a poll later, keep the last event's `seq` and pass it as
514
+ * { cursor } — the trace's cursor IS its position in the seq timeline.
515
+ */
516
+ type TraceEventPage = Page<TraceEvent>;
517
+ /** How many of the trials behind an aggregate were SCORED (means cover SCORED only). */
518
+ interface CompareCoverage {
519
+ scored: number;
520
+ total: number;
521
+ }
522
+ /**
523
+ * One (task, job) cell. `status` is a TrialStatus when every trial in the cell
524
+ * shares it, "MIXED" when they differ, "MISSING" when the job has no trials
525
+ * for the task.
526
+ */
527
+ interface CompareCell {
528
+ job_id: string;
529
+ status: TrialStatus | "MIXED" | "MISSING";
530
+ /** Mean reward over the cell's SCORED trials; null when none. Zero is a reward. */
531
+ mean_reward: number | null;
532
+ coverage: CompareCoverage;
533
+ }
534
+ /** One matrix row of jobs().compare(): a task across the compared jobs */
535
+ interface CompareTaskRow {
536
+ task_name: string;
537
+ /** True when the jobs' cells differ in status or reward for this task */
538
+ disagreement: boolean;
539
+ /** Cells in the caller's job-id order */
540
+ cells: CompareCell[];
541
+ }
542
+ /** Per-job aggregate of jobs().compare() */
543
+ interface CompareJobAggregate {
544
+ id: string;
545
+ datasets: DatasetRef[];
546
+ status: JobStatus;
547
+ /** Mean reward over SCORED trials only; null when none. Zero is a reward. */
548
+ mean_reward: number | null;
549
+ coverage: CompareCoverage;
550
+ cost_usd: number;
551
+ agents: AgentArm[];
552
+ started_at: string;
553
+ }
554
+ /**
555
+ * Result of jobs().compare([ids]): per-job aggregates plus a per-task matrix
556
+ * (disagreement rows first). `taskMatrix` is a frozen camelCase wire key.
557
+ */
558
+ interface CompareResponse {
559
+ /** Aggregates in the caller's id order */
560
+ jobs: CompareJobAggregate[];
561
+ taskMatrix: CompareTaskRow[];
562
+ }
563
+ /** Fields every job event carries, whatever its type. */
564
+ interface JobEventBase {
565
+ /** Monotonic sequence number (SSE id; the Last-Event-ID resume position) */
566
+ seq: number;
567
+ }
568
+ /** The job's resolved creation inputs, echoed so a watcher that joined late knows what it is watching. */
569
+ interface JobCreatedData {
570
+ datasets: DatasetRef[];
571
+ task_count: number;
572
+ agents: AgentArm[];
573
+ n_attempts: number;
574
+ n_concurrent_trials: number;
575
+ max_trial_spend_usd: number;
576
+ sandbox_provider: EvalSandboxProvider;
577
+ trial_count: number;
578
+ }
579
+ interface JobCancellingData {
580
+ job_id: string;
581
+ /** Queued trials cancelled outright by the request */
582
+ cancelled_trials: number;
583
+ /** Trials still in flight, winding down before the job settles */
584
+ active_trials: number;
585
+ }
586
+ interface JobCancelledData {
587
+ job_id: string;
588
+ /** Total queued trials cancelled across the request and the settle */
589
+ cancelled_trials: number;
590
+ }
591
+ interface TrialRunningData {
592
+ trial_id: string;
593
+ task_name: string;
594
+ }
595
+ interface TrialScoringData {
596
+ trial_id: string;
597
+ /** Bytes of agent stdout retained for the failure detail */
598
+ captured_bytes?: number;
599
+ }
600
+ /**
601
+ * A mid-run spend sample landed on a still-live trial. Emitted only when the
602
+ * reading actually updated a RUNNING/SCORING row, so a poll that raced the
603
+ * settle never fires one. The token sums come from the same ledger aggregation
604
+ * that produced the money figure, present only when the sample carried them —
605
+ * an older event replays without them.
606
+ */
607
+ interface TrialSpendData {
608
+ trial_id: string;
609
+ task_name: string;
610
+ /** The same lagging lower bound as Trial.live_spent_usd — not the trial's cost */
611
+ live_spent_usd: number;
612
+ /** Input tokens so far; includes cache tokens */
613
+ n_input_tokens?: number;
614
+ /** Cached input tokens so far (a subset of n_input_tokens) */
615
+ n_cache_tokens?: number;
616
+ /** Output tokens so far */
617
+ n_output_tokens?: number;
618
+ }
619
+ /**
620
+ * A trial reached a terminal status. `reward` is present only on the scored
621
+ * path; `exception_type` only on a failure; `attempt_phase` appears when the
622
+ * settle happened mid-phase (worker death), which is exactly when knowing the
623
+ * phase is worth having.
624
+ */
625
+ interface TrialSettledData {
626
+ trial_id: string;
627
+ task_name: string;
628
+ status: TrialStatus;
629
+ /** Zero is a reward; absent means the trial did not score. */
630
+ reward?: number | null;
631
+ exception_type?: string;
632
+ attempt_phase?: AttemptPhase | null;
633
+ }
634
+ /**
635
+ * One server-sent event from jobs().watch(), as a DISCRIMINATED UNION on
636
+ * `type` and ONLY on `type`: several event types carry identically shaped
637
+ * payloads (`job.running` and `job.completed` are both `{job_id}`), so payload
638
+ * shape can never route a reader — the `type` constant does. Switching on
639
+ * `type` narrows `data`.
640
+ *
641
+ * job.failed is declared terminal by the event stream and by this SDK, but NO
642
+ * SERVER PATH EMITS IT today. It stays in the union because both consumers
643
+ * treat it as terminal — the payload is fixed now so a client written today
644
+ * parses it when it first appears. Treat it as RESERVED rather than expected.
645
+ */
646
+ type JobEvent = (JobEventBase & {
647
+ type: "job.created";
648
+ data: JobCreatedData;
649
+ }) | (JobEventBase & {
650
+ type: "job.running";
651
+ data: {
652
+ job_id: string;
653
+ };
654
+ }) | (JobEventBase & {
655
+ type: "job.cancelling";
656
+ data: JobCancellingData;
657
+ }) | (JobEventBase & {
658
+ type: "job.cancelled";
659
+ data: JobCancelledData;
660
+ }) | (JobEventBase & {
661
+ type: "job.completed";
662
+ data: {
663
+ job_id: string;
664
+ };
665
+ }) | (JobEventBase & {
666
+ type: "job.failed";
667
+ data: {
668
+ job_id: string;
669
+ };
670
+ }) | (JobEventBase & {
671
+ type: "trial.running";
672
+ data: TrialRunningData;
673
+ }) | (JobEventBase & {
674
+ type: "trial.scoring";
675
+ data: TrialScoringData;
676
+ }) | (JobEventBase & {
677
+ type: "trial.spend";
678
+ data: TrialSpendData;
679
+ }) | (JobEventBase & {
680
+ type: "trial.settled";
681
+ data: TrialSettledData;
682
+ });
683
+ /**
684
+ * The handle returned by jobs().watch(). It is both:
685
+ * - a promise for the final Job — `await client.watch(id)` resolves once
686
+ * the job reaches a terminal status (the original form); and
687
+ * - an async iterable of events — `for await (const event of client.watch(id))`
688
+ * yields each JobEvent and completes on the terminal event.
689
+ *
690
+ * Pick one form per call: both drive the same underlying SSE stream, so a
691
+ * single handle should not be awaited and iterated at once.
692
+ */
693
+ interface JobWatch extends Awaitable<Job>, AsyncIterable<JobEvent> {
694
+ }
695
+ /** Cursor page of jobs (newest first) */
696
+ type JobPage = Page<Job>;
697
+ /**
698
+ * The handle returned by jobs().list(). Both:
699
+ * - a promise for a single JobPage — `await client.list({ limit })`
700
+ * returns one page (the original form); and
701
+ * - an async iterable — `for await (const item of client.list())` walks every
702
+ * job across cursor pages, fetching the next page for you.
703
+ */
704
+ interface JobList extends Awaitable<JobPage>, AsyncIterable<Job> {
705
+ }
706
+ /**
707
+ * One task's rollup within a job: its trial tally, mean reward over SCORED
708
+ * trials, and measured cost. Sits between the job body and the trial list so
709
+ * a caller need not fetch every trial to see which tasks are dragging.
710
+ */
711
+ interface JobTaskRollup {
712
+ task_name: string;
713
+ /** The dataset the task came from. */
714
+ source: string;
715
+ trials: TrialStatusTally;
716
+ /** Mean over SCORED trials only; null when none. Zero is a reward. */
717
+ mean_reward: number | null;
718
+ /** Measured spend across the task's settled trials. */
719
+ cost_usd: number | null;
720
+ }
721
+ /** Cursor page of per-task rollups */
722
+ type JobTaskRollupPage = Page<JobTaskRollup>;
723
+ /**
724
+ * The handle returned by jobs().tasks(). Both a promise for one page and an
725
+ * async iterable across cursor pages, like every other list handle.
726
+ */
727
+ interface JobTaskRollupList extends Awaitable<JobTaskRollupPage>, AsyncIterable<JobTaskRollup> {
728
+ }
729
+ /** Cursor page of trials */
730
+ type TrialPage = Page<Trial>;
731
+ /**
732
+ * The handle returned by jobs().trials(). Both:
733
+ * - a promise for a single TrialPage — `await client.trials(id, { limit })`
734
+ * returns one page (the original form); and
735
+ * - an async iterable — `for await (const trial of client.trials(id))` walks
736
+ * every trial across cursor pages, fetching the next page for you.
737
+ */
738
+ interface TrialList extends Awaitable<TrialPage>, AsyncIterable<Trial> {
739
+ }
740
+ /** Dataset version lifecycle state (wire values). Terminal: READY, FAILED, ARCHIVED. */
741
+ type DatasetVersionState = "DRAFT" | "IMPORTING" | "BUILDING" | "VALIDATING" | "READY" | "FAILED" | "ARCHIVED";
742
+ /** One immutable version of a dataset — one shape on every surface */
743
+ interface DatasetVersion {
744
+ version: string;
745
+ state: DatasetVersionState;
746
+ created_at: string;
747
+ task_count: number;
748
+ }
749
+ /**
750
+ * One provider's verdict for a task: runnable there, or refused with the
751
+ * limitation named (e.g. a multi-container task on a provider that cannot
752
+ * host its services, or declared resources above the provider's ceiling).
753
+ */
754
+ type TaskProviderVerdict = {
755
+ ok: true;
756
+ } | {
757
+ ok: false;
758
+ reason: string;
759
+ };
760
+ /** Public task fields only — instructions, environments, and tests never leave the server */
761
+ interface Task {
762
+ task_name: string;
763
+ agent_timeout_sec: number;
764
+ verifier_timeout_sec: number;
765
+ /**
766
+ * Where the task can run, per sandbox provider. Advisory for planning a
767
+ * job's provider choice — creating a job whose tasks include one refused on
768
+ * the chosen provider is rejected with the same reason, so nothing is ever
769
+ * spent on a trial that cannot execute.
770
+ */
771
+ providers: Record<EvalSandboxProvider, TaskProviderVerdict>;
772
+ }
773
+ /**
774
+ * Where a dataset's git source points now, versus what its active version was
775
+ * built from — the data behind a "new version available" badge. Null on a
776
+ * dataset whose source cannot be re-resolved; null is "nothing to watch",
777
+ * never "up to date". Nothing here imports anything — a new version is always
778
+ * a row you create (or `auto_import` creates).
779
+ */
780
+ interface UpstreamStatus {
781
+ /** The ref the active version was imported from. */
782
+ ref: string;
783
+ /** The commit the active version was built from. */
784
+ current_commit: string;
785
+ /** Where the ref points upstream now; null when the last check failed. */
786
+ latest_commit: string | null;
787
+ /** True when upstream has moved off the built-from commit. Branch on this. */
788
+ moved: boolean;
789
+ /** Reserved; always null today. */
790
+ behind_by: number | null;
791
+ /** When the cached answer was taken; null before the first check. */
792
+ checked_at: string | null;
793
+ /** Why the last check failed. Show "could not check", not "up to date". */
794
+ error: string | null;
795
+ /** Whether a moved upstream automatically imports a new version. */
796
+ auto_import: boolean;
797
+ }
798
+ /**
799
+ * A dataset in the catalog.
800
+ *
801
+ * list() returns the summary fields; get() additionally populates versions,
802
+ * selected_version, tasks, created_at, and updated_at.
803
+ */
804
+ interface Dataset {
805
+ name: string;
806
+ title: string | null;
807
+ description: string | null;
808
+ /** The active version, or null when none is active (bare-name job refs refuse). */
809
+ active_version: DatasetVersion | null;
810
+ /** All versions, newest first (get() only) */
811
+ versions?: DatasetVersion[];
812
+ /** The version whose tasks are listed below (get() only) */
813
+ selected_version?: DatasetVersion | null;
814
+ /**
815
+ * One page of the selected version's tasks (get() only). Paged like every
816
+ * other collection: a SWE-bench-scale dataset has thousands of tasks, so
817
+ * pass { limit, cursor } to get() and follow nextCursor.
818
+ */
819
+ tasks?: Page<Task>;
820
+ upstream: UpstreamStatus | null;
821
+ /** get() only */
822
+ created_at?: string;
823
+ /** get() only */
824
+ updated_at?: string;
825
+ }
826
+ /** Body of datasets().update() — the only settable dataset field. */
827
+ interface DatasetPatch {
828
+ /** Automatically import a new version when the upstream git ref moves. */
829
+ upstream_auto_import: boolean;
830
+ }
831
+ /**
832
+ * A dataset's active version resolved to a runnable shape.
833
+ *
834
+ * Unlike Dataset, `version` and `tasks` are non-optional: datasets()
835
+ * .getActive() throws NoActiveVersionError when there is no active version,
836
+ * so callers never branch on a missing active version.
837
+ */
838
+ interface ActiveDataset {
839
+ name: string;
840
+ title: string | null;
841
+ description: string | null;
842
+ /** The active version (always present) */
843
+ active_version: DatasetVersion;
844
+ /** The active version string (identical to active_version.version) */
845
+ version: string;
846
+ /** One page of the active version's tasks */
847
+ tasks: Page<Task>;
848
+ /** All versions, newest first */
849
+ versions: DatasetVersion[];
850
+ created_at: string;
851
+ updated_at: string;
852
+ }
853
+ /** Cursor page of datasets */
854
+ type DatasetPage = Page<Dataset>;
855
+ /** Dual-use handle from datasets().list(): await one page, or iterate them all */
856
+ interface DatasetList extends Awaitable<DatasetPage>, AsyncIterable<Dataset> {
857
+ }
858
+ /**
859
+ * Source for datasets().publish(): EITHER a git repository pinned to a ref, OR
860
+ * a local corpus directory (tarred deterministically on the client and
861
+ * uploaded).
862
+ *
863
+ * A UNION, not three optional fields: `{}` and both-branches-at-once are
864
+ * compile errors rather than a 400 the caller discovers at run time, and
865
+ * `?: never` on the absent branch's keys is what rejects the excess property
866
+ * through a variable. `git_ref` is REQUIRED on the git branch — an unpinned
867
+ * import is not reproducible.
868
+ */
869
+ type DatasetSource = {
870
+ /**
871
+ * A git repository URL. https:// only — the import runs on a worker with
872
+ * no ssh client, so ssh:// and git@ remotes are refused at validation.
873
+ * For a private repository, put a token in the https url.
874
+ */
875
+ git_url: string;
876
+ /** A pinned branch, tag, or commit. Required: an unpinned import is not reproducible. */
877
+ git_ref: string;
878
+ directory?: never;
879
+ } | {
880
+ /** A local standard-layout corpus directory — tarred + gzipped and uploaded. */
881
+ directory: string;
882
+ git_url?: never;
883
+ git_ref?: never;
884
+ };
885
+ /** Input for datasets().publish() */
886
+ interface PublishDatasetInput {
887
+ source: DatasetSource;
888
+ /** Catalog dataset name the version lands under (created or extended) */
889
+ name: string;
890
+ /** Version label for the new immutable version */
891
+ version: string;
892
+ }
893
+ /**
894
+ * Dataset import status.
895
+ *
896
+ * The SAME four words a job uses, deliberately: a status chip rendering both
897
+ * never carries a translation table. Terminal: "COMPLETED" (the corpus landed
898
+ * as a dataset version; runnable once activated) and "FAILED".
899
+ */
900
+ type DatasetImportStatus = "QUEUED" | "RUNNING" | "COMPLETED" | "FAILED";
901
+ /** Structured failure detail for a FAILED import. */
902
+ interface DatasetImportFailure {
903
+ /** Stable machine-readable cause; "import_failed" when none was recorded. */
904
+ code: string;
905
+ /** What went wrong, e.g. "2/113 task(s) failed to parse" */
906
+ message: string;
907
+ /** Per-task parse/validation failures, when the corpus was reachable */
908
+ failures?: {
909
+ task_name: string;
910
+ error: string;
911
+ }[];
912
+ }
913
+ /**
914
+ * Non-fatal but consequential import outcome. A version whose warnings include
915
+ * `no_solutions_archived` cannot be activated through this API
916
+ * (`version_not_activatable`) — an import that will never become runnable must
917
+ * not look identical to one that will.
918
+ */
919
+ interface ImportWarning {
920
+ code: "solutions_archiving_disabled" | "no_solutions_archived" | "partial_solutions_archived";
921
+ message?: string;
922
+ }
923
+ /**
924
+ * An asynchronous publish. Self-describing: every response names the
925
+ * dataset@version being imported — the 202 from publish(), getImport(), and
926
+ * listImports() all return this same shape, so a caller can render the row it
927
+ * just created without a follow-up read.
928
+ */
929
+ interface DatasetImport {
930
+ /** Import job id */
931
+ id: string;
932
+ status: DatasetImportStatus;
933
+ /** Catalog dataset name the import creates or extends */
934
+ name: string;
935
+ /** Version label of the imported version */
936
+ version: string;
937
+ /**
938
+ * Why the import FAILED; null otherwise. Named `failure`, never `error` —
939
+ * see JobFailure.
940
+ */
941
+ failure: DatasetImportFailure | null;
942
+ /** Non-fatal but consequential outcomes — see ImportWarning. */
943
+ warnings: ImportWarning[];
944
+ /** Number of tasks parsed, once counted */
945
+ task_count?: number;
946
+ created_at?: string;
947
+ updated_at?: string;
948
+ }
949
+ /** Cursor page of dataset imports */
950
+ type DatasetImportPage = Page<DatasetImport>;
951
+ /** Dual-use handle from datasets().listImports(): await one page, or iterate them all */
952
+ interface DatasetImportList extends Awaitable<DatasetImportPage>, AsyncIterable<DatasetImport> {
953
+ }
954
+ /**
955
+ * Where a registered agent's executables came from: an install script run in a
956
+ * throwaway builder sandbox, or a tarball uploaded from a local directory.
957
+ * Echoed on every response; the SDK never guesses it.
958
+ */
959
+ type AgentSource = "install_script" | "tarball";
960
+ /**
961
+ * A private agent registered by the caller. Once registered, its `name` is
962
+ * usable in job `agents[].name` exactly like a built-in ("claude", "codex",
963
+ * ...). Private to its owner: another user's name reads as `agent_not_found`,
964
+ * never as a permission error — existence is never leaked.
965
+ */
966
+ interface Agent {
967
+ /** The name to put in job agents[].name */
968
+ name: string;
969
+ /** How the executables were produced */
970
+ source: AgentSource;
971
+ /** The command run headless with `sh -c` at the task working directory */
972
+ run_command: string;
973
+ /**
974
+ * Caller-declared env injected at RUN time only. It may not override the run
975
+ * contract's own keys — the server rejects that at registration with
976
+ * `agent_invalid_env`.
977
+ */
978
+ env: Record<string, string>;
979
+ created_at: string;
980
+ updated_at: string;
981
+ }
982
+ /**
983
+ * The two sources a registered agent's executables can come from. A union, not
984
+ * two optional fields — see DatasetSource for why `?: never` is load-bearing
985
+ * rather than decorative.
986
+ */
987
+ type AgentSourceInput = {
988
+ /**
989
+ * The install script itself (not a path). It runs in a throwaway builder
990
+ * sandbox that has internet and ZERO secrets, so everything it fetches
991
+ * must be publicly fetchable, and it must leave executables in
992
+ * `$PREFIX/bin`.
993
+ */
994
+ install_script: string;
995
+ directory?: never;
996
+ } | {
997
+ /**
998
+ * A local directory holding the agent — tarred + gzipped and uploaded.
999
+ * Same build rules as an install script.
1000
+ */
1001
+ directory: string;
1002
+ install_script?: never;
1003
+ };
1004
+ /**
1005
+ * Input for agents().create(): a name, a run command, and EXACTLY ONE source.
1006
+ * The source half is a union, so omitting both or passing both is a compile
1007
+ * error rather than a 400 the caller discovers at run time.
1008
+ */
1009
+ type AgentInput = AgentSourceInput & {
1010
+ /** Agent name; also the value used later in job agents[].name */
1011
+ name: string;
1012
+ /** Command run headless with `sh -c` at the task working directory */
1013
+ run_command: string;
1014
+ /** Env injected at RUN time only; may not override the run contract's keys */
1015
+ env?: Record<string, string>;
1016
+ };
1017
+ /**
1018
+ * An agent upsert body. Same shape as AgentInput minus `name`, which the
1019
+ * upsert takes as its first argument — the name is the resource identity, not
1020
+ * a field of it.
1021
+ */
1022
+ type AgentUpsertInput = AgentSourceInput & {
1023
+ /** Command run headless with `sh -c` at the task working directory */
1024
+ run_command: string;
1025
+ /** Env injected at RUN time only; may not override the run contract's keys */
1026
+ env?: Record<string, string>;
1027
+ };
1028
+ /** Cursor page of registered agents */
1029
+ type AgentPage = Page<Agent>;
1030
+ /** Dual-use handle from agents().list(): await one page, or iterate them all */
1031
+ interface AgentList extends Awaitable<AgentPage>, AsyncIterable<Agent> {
1032
+ }
1033
+ /** Options for jobs().start() and resume() */
1034
+ interface StartJobOptions {
1035
+ /**
1036
+ * Idempotency-Key header value: retries with the same key return the
1037
+ * original job (idempotent_replay: true) instead of creating a new one.
1038
+ */
1039
+ idempotencyKey?: string;
1040
+ }
1041
+ /** Options for jobs().list() (default page 50, max 200) */
1042
+ interface ListJobsOptions extends PageOptions {
1043
+ /** Server-side free-text filter over job name and dataset names. */
1044
+ search?: string;
1045
+ }
1046
+ /** Options for jobs().tasks() (default page 50, max 200) */
1047
+ interface ListJobTasksOptions extends PageOptions {
1048
+ }
1049
+ /** Options for jobs().trials() (default page 50, max 200) */
1050
+ interface ListTrialsOptions extends PageOptions {
1051
+ /** Only trials in these statuses (e.g. the failures behind a resume decision) */
1052
+ status?: TrialStatus[];
1053
+ /** Only one dataset's trials — exact match on the trial's `source`. */
1054
+ dataset?: string;
1055
+ }
1056
+ /** Options for datasets().list() (default page 50, max 200) */
1057
+ interface ListDatasetsOptions extends PageOptions {
1058
+ /** Server-side free-text filter over name and description. */
1059
+ search?: string;
1060
+ }
1061
+ /** Options for agents().list() (default page 50, max 200) */
1062
+ interface ListAgentsOptions extends PageOptions {
1063
+ }
1064
+ /** Options for datasets().get() / getActive(): pages the TASK list (default 200, max 500) */
1065
+ interface GetDatasetOptions extends PageOptions {
1066
+ }
1067
+ /** Options for datasets().listImports() */
1068
+ interface ListImportsOptions extends PageOptions {
1069
+ /** Only imports in this status */
1070
+ status?: DatasetImportStatus;
1071
+ /** Only imports of this dataset name */
1072
+ dataset?: string;
1073
+ }
1074
+ /** Options for trials().trace() and traceEvents() */
1075
+ interface TraceOptions extends PageOptions {
1076
+ /**
1077
+ * Resume position: events with seq strictly greater than this cursor (omit =
1078
+ * from the beginning). A trace cursor IS a seq, so to resume a poll later
1079
+ * pass the last event's `seq` here as a string.
1080
+ */
1081
+ cursor?: string;
1082
+ /** Max events per page (server default: 200, max: 1000) */
1083
+ limit?: number;
1084
+ }
1085
+ /** Options for datasets().watchImport() */
1086
+ interface WatchImportOptions {
1087
+ /** Called on every observed status change (including the first status seen) */
1088
+ onStatus?: (datasetImport: DatasetImport) => void;
1089
+ /** Abort the watch (rejects with the abort reason) */
1090
+ signal?: AbortSignal;
1091
+ /** Poll interval between getImport() calls (default: 2000ms) */
1092
+ pollIntervalMs?: number;
1093
+ }
1094
+ /** Options for jobs().watch() */
1095
+ interface WatchJobOptions {
1096
+ /** Called for every event (replayed + live) */
1097
+ onEvent?: (event: JobEvent) => void;
1098
+ /** Abort the watch (rejects with the abort reason) */
1099
+ signal?: AbortSignal;
1100
+ /** Initial reconnect backoff (default: 1000ms; doubles up to maxReconnectDelayMs) */
1101
+ reconnectDelayMs?: number;
1102
+ /** Backoff ceiling (default: 30000ms) */
1103
+ maxReconnectDelayMs?: number;
1104
+ }
1105
+ /** Delivery options for datasets().download() */
1106
+ interface DownloadDatasetOptions {
1107
+ /** Directory to save the package into (returns the file path) */
1108
+ to?: string;
1109
+ /** Return the raw response stream instead of a Buffer */
1110
+ stream?: boolean;
1111
+ }
1112
+ /** Delivery options for jobs().download() */
1113
+ interface DownloadJobOptions {
1114
+ /** Directory to save the archive into (returns the file path) */
1115
+ to?: string;
1116
+ /** Return the raw response stream instead of a Buffer */
1117
+ stream?: boolean;
1118
+ }
1119
+ /** Client for the shared dataset catalog */
1120
+ interface DatasetsClient {
1121
+ /**
1122
+ * List datasets with their active versions (cursor-paged). Await the
1123
+ * result for one page, or `for await` it to walk the whole catalog.
1124
+ */
1125
+ list(options?: ListDatasetsOptions): DatasetList;
1126
+ /**
1127
+ * Get one dataset: all versions + one page of the selected version's tasks.
1128
+ * ref is "name" (active version's tasks) or "name@version"; { limit, cursor }
1129
+ * page the tasks.
1130
+ */
1131
+ get(ref: string, options?: GetDatasetOptions): Promise<Dataset>;
1132
+ /**
1133
+ * Get a dataset's active version resolved to a runnable shape: unlike
1134
+ * get(), `version` and `tasks` are guaranteed present. Throws
1135
+ * NoActiveVersionError when the dataset has no active version. Use get()
1136
+ * for the full multi-version detail with optional fields.
1137
+ */
1138
+ getActive(name: string, options?: GetDatasetOptions): Promise<ActiveDataset>;
1139
+ /**
1140
+ * Publish a dataset version (asynchronous server-side import) from a git
1141
+ * source pinned to a ref, or a local corpus directory. Returns immediately;
1142
+ * poll with getImport()/watchImport().
1143
+ */
1144
+ publish(input: PublishDatasetInput): Promise<DatasetImport>;
1145
+ /** Get an import job's status (failure, warnings, and task_count when available) */
1146
+ getImport(id: string): Promise<DatasetImport>;
1147
+ /**
1148
+ * Poll getImport() until the import reaches a terminal status ("COMPLETED"
1149
+ * or "FAILED") and resolve with the final import.
1150
+ */
1151
+ watchImport(id: string, options?: WatchImportOptions): Promise<DatasetImport>;
1152
+ /**
1153
+ * List the caller's own imports, newest first (cursor-paged). This is how
1154
+ * you find an import again after losing the id publish() returned. Await
1155
+ * for one page, or `for await` to walk them all. { status } filters on the
1156
+ * import vocabulary; { dataset } narrows to one dataset name.
1157
+ */
1158
+ listImports(options?: ListImportsOptions): DatasetImportList;
1159
+ /**
1160
+ * Download the ORIGINAL corpus package one of your own dataset versions was
1161
+ * published from — the gzipped tarball you uploaded, or, for a git publish,
1162
+ * the checked-out tree packed at import time. `ref` is "name" (the active
1163
+ * version's package) or "name@version".
1164
+ *
1165
+ * OWNER ONLY. This is the one call that returns task files, and it returns
1166
+ * them only to the account that owns the dataset; a platform-curated dataset
1167
+ * has no owner, so nobody can download it. Someone else's dataset is a plain
1168
+ * not-found, never a 403.
1169
+ *
1170
+ * The server verifies the stored bytes against their recorded sha256 before
1171
+ * sending anything and echoes the digest; the client re-checks the digest
1172
+ * and the Content-Length, so a successful call is byte-identical to what was
1173
+ * published. A version published before packages were retained has none
1174
+ * (`package_not_retained`, distinct from "not found").
1175
+ *
1176
+ * Default: Buffer. { to } saves into a directory and returns the file path.
1177
+ * { stream: true } returns the raw response stream.
1178
+ */
1179
+ download(ref: string): Promise<Buffer>;
1180
+ download(ref: string, options: {
1181
+ to: string;
1182
+ }): Promise<string>;
1183
+ download(ref: string, options: {
1184
+ stream: true;
1185
+ }): Promise<ReadableStream<Uint8Array>>;
1186
+ download(ref: string, options?: DownloadDatasetOptions): Promise<Buffer | string | ReadableStream<Uint8Array>>;
1187
+ /**
1188
+ * Update dataset settings. The only settable field is
1189
+ * `upstream_auto_import`: automatically import a new version when the
1190
+ * dataset's upstream git ref moves. Refused (upstream_not_watchable) when
1191
+ * the dataset has no moving git ref to follow, and dataset_not_owned on a
1192
+ * platform-curated dataset. Returns the updated dataset.
1193
+ */
1194
+ update(name: string, patch: DatasetPatch): Promise<Dataset>;
1195
+ /**
1196
+ * Activate a READY version you own: bare-name job references resolve to it
1197
+ * from then on. Refused with `version_not_ready` while the import still
1198
+ * runs and `version_not_activatable` for a version that can never be
1199
+ * activated (for example, no reference solutions were archived).
1200
+ */
1201
+ activate(name: string, version: string): Promise<Dataset>;
1202
+ /**
1203
+ * Delete a dataset you own, with every version, task, and archived
1204
+ * solution. Refused (dataset_in_use) while any job still references it — a
1205
+ * dataset is never deleted out from under a job that measured against it,
1206
+ * and `err.details.sampleJobIds` names the jobs blocking it. A platform
1207
+ * dataset is refused with dataset_not_owned; a name you cannot see is a
1208
+ * plain not-found.
1209
+ */
1210
+ delete(name: string): Promise<void>;
1211
+ }
1212
+ /** Client for the caller's own private (bring-your-own) agents */
1213
+ interface AgentsClient {
1214
+ /**
1215
+ * Register a private agent. Provide either an install script
1216
+ * (`{ install_script }`) or a local directory (`{ directory }`), never both.
1217
+ * The name is then usable in job `agents[].name` like a built-in.
1218
+ */
1219
+ create(input: AgentInput): Promise<Agent>;
1220
+ /**
1221
+ * List the caller's registered agents (cursor-paged). Await the result for
1222
+ * one page, or `for await` it to walk them all.
1223
+ */
1224
+ list(options?: ListAgentsOptions): AgentList;
1225
+ /** Get one registered agent by name */
1226
+ get(name: string): Promise<Agent>;
1227
+ /** Delete a registered agent. Past jobs keep their recorded agent. */
1228
+ delete(name: string): Promise<void>;
1229
+ /**
1230
+ * Register or replace an agent in ONE call, under the name you give.
1231
+ *
1232
+ * Use this instead of delete()+create() to change an existing registration:
1233
+ * the pair leaves a window where the agent does not exist, and anything
1234
+ * naming it in that window fails for a change that was only ever meant to be
1235
+ * an edit. This is a full replacement, not a patch — every field comes from
1236
+ * this call, and an omitted `env` becomes empty.
1237
+ */
1238
+ upsert(name: string, input: AgentUpsertInput): Promise<Agent>;
1239
+ }
1240
+ /** Client for hosted jobs */
1241
+ interface JobsClient {
1242
+ /**
1243
+ * Start a job over one or more catalog datasets. Each dataset selector may
1244
+ * carry glob task filters; every agent arm must name a model. Supports
1245
+ * Idempotency-Key.
1246
+ */
1247
+ start(input: JobCreate, options?: StartJobOptions): Promise<Job>;
1248
+ /** Get one job */
1249
+ get(id: string): Promise<Job>;
1250
+ /**
1251
+ * List the caller's jobs, newest first (cursor-paged). Await the
1252
+ * result for one page, or `for await` it to walk every job across
1253
+ * cursor pages transparently.
1254
+ */
1255
+ list(options?: ListJobsOptions): JobList;
1256
+ /**
1257
+ * List a job's trials (cursor-paged; { status } filters, e.g. to
1258
+ * the failed trials). Await the result for one page, or `for await` it to
1259
+ * walk every trial across cursor pages transparently.
1260
+ */
1261
+ trials(id: string, options?: ListTrialsOptions): TrialList;
1262
+ /**
1263
+ * Per-task rollup of a job (cursor-paged): one row per distinct task with
1264
+ * its trial tally, mean reward, and cost. Sits between the job body and
1265
+ * the trial list so a caller need not fetch every trial to see which
1266
+ * tasks are dragging.
1267
+ */
1268
+ tasks(id: string, options?: ListJobTasksOptions): JobTaskRollupList;
1269
+ /**
1270
+ * Watch a job's event stream (SSE). Replays from the beginning,
1271
+ * resumes with Last-Event-ID on reconnect (exponential backoff), and
1272
+ * finishes on the terminal event.
1273
+ *
1274
+ * The returned handle is dual-use: `await client.watch(id)` resolves with the
1275
+ * final Job, or `for await (const event of client.watch(id))` iterates
1276
+ * the events. The `onEvent` callback still fires in both forms.
1277
+ */
1278
+ watch(id: string, options?: WatchJobOptions): JobWatch;
1279
+ /** Request cancellation. Idempotent; a terminal job is a no-op. */
1280
+ cancel(id: string): Promise<Job>;
1281
+ /**
1282
+ * Resume a terminal job: a NEW linked job holding fresh trials for the
1283
+ * source's failed and stopped work (`source_jobs` records
1284
+ * `action: "resume"`); the source is never mutated.
1285
+ * `request.filter_error_types` selects which failures to resume by
1286
+ * `exception_info.exception_type`; omitted, the default set includes
1287
+ * stopped trials. Supports Idempotency-Key.
1288
+ */
1289
+ resume(id: string, request?: ResumeRequest, options?: StartJobOptions): Promise<Job>;
1290
+ /**
1291
+ * Regrade a terminal job: re-run the verifier of every REGRADABLE trial
1292
+ * against its recorded inputs, in fresh separate verifier boxes. The agent
1293
+ * phase is never re-run and the source trials are never modified. THE
1294
+ * RESPONSE IS A JOB — a regrade is an ordinary job whose `source_jobs`
1295
+ * records `action: "regrade"` and whose `is_regrade` is true; view it with
1296
+ * get(). `request` narrows the set by statuses and/or task.
1297
+ */
1298
+ regrade(id: string, request?: RegradeRequest): Promise<Job>;
1299
+ /**
1300
+ * Side-by-side comparison of 2-10 owned jobs: per-job
1301
+ * aggregates plus a per-task matrix with disagreement rows first.
1302
+ */
1303
+ compare(ids: string[]): Promise<CompareResponse>;
1304
+ /**
1305
+ * Download a terminal job's results archive (gzipped, standard results
1306
+ * layout, deterministic bytes). Default: Buffer — verified against the
1307
+ * response's Content-Length and, when the server states one, its digest.
1308
+ * { to } saves to a directory (temp-then-rename, same verification) and
1309
+ * returns the file path. { stream: true } returns the raw response stream,
1310
+ * the one shape the caller must verify themselves.
1311
+ */
1312
+ download(id: string): Promise<Buffer>;
1313
+ download(id: string, options: {
1314
+ to: string;
1315
+ }): Promise<string>;
1316
+ download(id: string, options: {
1317
+ stream: true;
1318
+ }): Promise<ReadableStream<Uint8Array>>;
1319
+ download(id: string, options?: DownloadJobOptions): Promise<Buffer | string | ReadableStream<Uint8Array>>;
1320
+ }
1321
+ /**
1322
+ * The trace route's `?stream=` selectors, in the contract's own order —
1323
+ * `trace-parsed` (the parsed event trace, the same answer as omitting
1324
+ * `stream`) followed by the raw-artifact vocabulary. A runtime value (not
1325
+ * only a type) so a drift gate can hold it to the spec's enum, and the CLI
1326
+ * can build its `--stream` validation from the same list instead of a
1327
+ * second copy.
1328
+ */
1329
+ declare const TRIAL_ARTIFACT_STREAMS: readonly ["trace-parsed", "verifier", "trace-stdout", "trace-stderr", "trajectory", "agent-home"];
1330
+ /** One `?stream=` selector on the trace route. */
1331
+ type TrialArtifactStream = (typeof TRIAL_ARTIFACT_STREAMS)[number];
1332
+ /** Client for globally addressable trials — no job id in any signature */
1333
+ interface TrialsClient {
1334
+ /**
1335
+ * Get one trial by its globally addressable id. The body carries `job_id`
1336
+ * as the reverse pointer; `exception_info.exception_message` is untruncated
1337
+ * here, unlike list rows.
1338
+ */
1339
+ get(trialId: string): Promise<Trial>;
1340
+ /** Get one page of a trial's trace; resume with { cursor: page.nextCursor } */
1341
+ trace(trialId: string, options?: TraceOptions): Promise<TraceEventPage>;
1342
+ /**
1343
+ * Iterate a trial's trace events, fetching pages under the hood until
1344
+ * the currently available trace is drained. Resume later by passing the
1345
+ * last seen seq as { cursor }.
1346
+ */
1347
+ traceEvents(trialId: string, options?: TraceOptions): AsyncIterableIterator<TraceEvent>;
1348
+ /**
1349
+ * One RAW trace artifact, by the trace route's ?stream= selector.
1350
+ * "verifier" | "trace-stdout" | "trace-stderr" answer the log text;
1351
+ * "agent-home" answers the CLI's whole home folder (subagent transcripts
1352
+ * included), keyed by sandbox path. Null = never stored
1353
+ * (a normal answer, not an error). "trajectory" is in the vocabulary ahead
1354
+ * of its server wave — until that wave lands the route answers not-found,
1355
+ * reported honestly as the API error it is. "trace-parsed" is not an
1356
+ * artifact — the parsed event trace rides trace()/traceEvents().
1357
+ */
1358
+ artifact(trialId: string, stream: Exclude<TrialArtifactStream, "trace-parsed" | "agent-home">): Promise<string | null>;
1359
+ artifact(trialId: string, stream: "agent-home"): Promise<Record<string, string> | null>;
1360
+ /**
1361
+ * Regrade one settled trial: re-run its verifier against its recorded
1362
+ * inputs in a fresh separate verifier box. Refused
1363
+ * (regrade_source_ineligible) for shared-mode or pre-persistence trials.
1364
+ * THE RESPONSE IS A JOB — a one-trial regrade job with `source_jobs`
1365
+ * recording the provenance.
1366
+ */
1367
+ regrade(trialId: string): Promise<Job>;
1368
+ /**
1369
+ * Stop selected in-flight trials without cancelling their job: each trial's
1370
+ * sandbox is killed and the trial is settled with its spend read from the
1371
+ * gateway. Only the caller's own trials; ids belonging to someone else are
1372
+ * reported in `not_found` (existence is never leaked). Idempotent —
1373
+ * already-terminal trials are reported as such and left untouched.
1374
+ */
1375
+ stop(trialIds: string[]): Promise<StopResponse>;
1376
+ }
1377
+ /** A key descriptor. The secret is never returned. */
1378
+ interface ApiKey {
1379
+ id: string;
1380
+ label: string | null;
1381
+ created_at: string;
1382
+ last_used_at: string | null;
1383
+ }
1384
+ /** Who the caller is and the key they used. */
1385
+ interface AuthStatus {
1386
+ user_id: string;
1387
+ email: string | null;
1388
+ key: ApiKey;
1389
+ }
1390
+ /** Client for caller identity. */
1391
+ interface AuthClient {
1392
+ /** Identify the caller and the API key in use. */
1393
+ status(): Promise<AuthStatus>;
1394
+ }
1395
+ /**
1396
+ * Every error code the hosted API can return, as a closed list.
1397
+ *
1398
+ * This exists so a typo cannot compile. `err.code === "insufficient_creidts"`
1399
+ * used to typecheck (code was `string`) and then silently never match, which is
1400
+ * the worst shape a bug can take: the branch looks handled and never runs.
1401
+ *
1402
+ * It mirrors the ErrorCode enum in spec/openapi.yaml and is published verbatim
1403
+ * at GET /api/meta as `error_codes`. A server newer than this SDK may send a
1404
+ * code that is not listed here — `EvolveApiError.code` widens to string for
1405
+ * exactly that case, so an unknown code is still readable, just not narrowable.
1406
+ *
1407
+ * Held to the spec by hosted-error-codes.json at the package root, the
1408
+ * checked-in copy both SDKs assert against; the list drifted silently before
1409
+ * that file existed. Adding a code means editing the spec, that file, this
1410
+ * list, and the Python pair.
1411
+ */
1412
+ declare const HOSTED_ERROR_CODES: readonly ["missing_authorization", "invalid_api_key", "credential_service_unavailable", "rate_limited", "insufficient_credits", "invalid_json", "invalid_input", "invalid_limit", "invalid_status", "invalid_cursor", "invalid_after", "invalid_format", "invalid_ids", "invalid_multipart", "idempotency_key_reused", "dataset_not_found", "dataset_version_not_found", "dataset_name_taken", "dataset_in_use", "dataset_not_owned", "upstream_not_watchable", "no_active_version", "version_not_ready", "version_not_activatable", "unknown_task_names", "no_tasks", "agent_not_found", "agent_name_taken", "agent_name_reserved", "agent_invalid_name", "agent_source_required", "agent_source_conflict", "agent_invalid_env", "agent_too_large", "agent_limit_reached", "agent_version_not_found", "job_too_large", "provider_unsupported", "job_not_found", "job_not_terminal", "no_failed_trials", "trial_not_found", "concurrent_update", "regrade_source_ineligible", "no_regradable_trials", "import_not_found", "import_too_large", "invalid_archive", "package_not_retained", "package_corrupt", "package_missing", "internal_error"];
1413
+ /** One of the API's stable error codes. */
1414
+ type HostedErrorCode = (typeof HOSTED_ERROR_CODES)[number];
1415
+ /** True when `value` is a code this SDK version knows about (narrowing guard). */
1416
+ declare function isHostedErrorCode(value: unknown): value is HostedErrorCode;
1417
+ /** A closed vocabulary a client renders, with the members that end it. */
1418
+ interface StatusVocabulary {
1419
+ values: string[];
1420
+ /** Members after which nothing more happens — a watcher may stop here. */
1421
+ terminal: string[];
1422
+ }
1423
+ /** One built-in agent's declared capabilities. */
1424
+ interface AgentCapability {
1425
+ name: string;
1426
+ /** Whether job agents[].reasoning_effort reaches this agent. */
1427
+ effort_support: boolean;
1428
+ /** Whether job agents[].version may pin this agent. */
1429
+ version_pinnable: boolean;
1430
+ /**
1431
+ * Newest published version, for a "your pin is out of date" badge. Null
1432
+ * means "not known right now", never "up to date".
1433
+ */
1434
+ latest_version?: string | null;
1435
+ }
1436
+ /** One sandbox provider, its ceilings, and what it refuses. */
1437
+ interface ProviderCapability {
1438
+ name: string;
1439
+ default: boolean;
1440
+ sizing: {
1441
+ max_cpus: number;
1442
+ max_memory_mb: number;
1443
+ max_storage_mb: number;
1444
+ storage: "sized" | "fixed";
1445
+ };
1446
+ refuses: {
1447
+ capability: string;
1448
+ reason: string;
1449
+ }[];
1450
+ }
1451
+ /**
1452
+ * One managed sandbox door and whether this deployment serves it — a
1453
+ * different question from ProviderCapability, which is about the eval lane.
1454
+ * A managed sandbox is one the caller drives directly holding nothing but an
1455
+ * Evolve key.
1456
+ */
1457
+ interface ManagedProviderCapability {
1458
+ name: string;
1459
+ /**
1460
+ * The operator config this door reads is present. NOT a health check: it
1461
+ * says nothing about whether the pass-through behind the door is deployed
1462
+ * or the credential behind it is valid.
1463
+ */
1464
+ configured: boolean;
1465
+ /** Config this door reads, so an operator sees what to set. */
1466
+ requires_config: string[];
1467
+ /** The subset of `requires_config` missing right now — empty when configured. */
1468
+ missing_config: string[];
1469
+ /** A full SDK agent session can run on this door. */
1470
+ agent_sessions: boolean;
1471
+ /** Why not, when `agent_sessions` is false. Null otherwise. */
1472
+ agent_sessions_reason: string | null;
1473
+ }
1474
+ /**
1475
+ * The capability document: everything a client would otherwise hardcode.
1476
+ *
1477
+ * Public and cacheable — no API key needed, so a signed-out page can populate
1478
+ * its own agent picker. `schema_version` bumps when a FIELD changes meaning,
1479
+ * never when a value changes.
1480
+ */
1481
+ interface CapabilityDocument {
1482
+ schema_version: number;
1483
+ /** Built-in agents and their declared capabilities. */
1484
+ agents: AgentCapability[];
1485
+ /** Rules a bring-your-own agent registration must satisfy. */
1486
+ agent_registration: {
1487
+ name_pattern: string;
1488
+ max_name_length: number;
1489
+ max_run_command_length: number;
1490
+ max_install_script_length: number;
1491
+ max_env_entries: number;
1492
+ max_per_user: number;
1493
+ max_upload_bytes: number;
1494
+ /** Built-in names a registration may not reuse. */
1495
+ reserved_names: string[];
1496
+ /** Env keys the platform owns; declaring one is refused at registration. */
1497
+ reserved_env_keys: string[];
1498
+ };
1499
+ sandbox_providers: ProviderCapability[];
1500
+ /** The managed doors this deployment serves, and what each can carry. */
1501
+ managed_providers: ManagedProviderCapability[];
1502
+ /** Constraints that hold on EVERY provider. */
1503
+ platform_constraints: {
1504
+ capability: string;
1505
+ reason: string;
1506
+ }[];
1507
+ network_modes: string[];
1508
+ /** Every status vocabulary on the surface, with terminal members. */
1509
+ statuses: {
1510
+ job: StatusVocabulary;
1511
+ trial: StatusVocabulary;
1512
+ import: StatusVocabulary;
1513
+ dataset_version: StatusVocabulary;
1514
+ };
1515
+ limits: {
1516
+ job: {
1517
+ max_n_attempts: number;
1518
+ max_agents: number;
1519
+ max_trials: number;
1520
+ n_concurrent_trials: {
1521
+ default: number;
1522
+ max: number;
1523
+ };
1524
+ default_max_trial_spend_usd: number;
1525
+ default_sandbox_provider: string;
1526
+ default_sizing: {
1527
+ cpus: number;
1528
+ memory_mb: number;
1529
+ storage_mb: number;
1530
+ };
1531
+ /** Every agent must name a model; the server applies no default. */
1532
+ model_required: boolean;
1533
+ /**
1534
+ * Phase wall-clocks a task INHERITS when its own config declares none —
1535
+ * a task that declares its own always wins, so these fill in rather than
1536
+ * cap. Published because nothing else here says how long a trial may run.
1537
+ */
1538
+ default_agent_timeout_sec: number;
1539
+ default_verifier_timeout_sec: number;
1540
+ /** Values agents[].reasoning_effort accepts, and the one an omitted effort takes. */
1541
+ reasoning_efforts: string[];
1542
+ default_reasoning_effort: string;
1543
+ };
1544
+ compare: {
1545
+ min_ids: number;
1546
+ max_ids: number;
1547
+ };
1548
+ pagination: {
1549
+ collections: {
1550
+ default: number;
1551
+ max: number;
1552
+ };
1553
+ dataset_tasks: {
1554
+ default: number;
1555
+ max: number;
1556
+ };
1557
+ trace_events: {
1558
+ default: number;
1559
+ max: number;
1560
+ };
1561
+ };
1562
+ uploads: {
1563
+ dataset_archive_bytes: number;
1564
+ agent_tarball_bytes: number;
1565
+ };
1566
+ dataset_names: {
1567
+ pattern: string;
1568
+ max_name_length: number;
1569
+ max_version_length: number;
1570
+ max_git_url_length: number;
1571
+ max_git_ref_length: number;
1572
+ };
1573
+ /** How many items an error MESSAGE names before "and N more". */
1574
+ max_items_named_in_error_message: number;
1575
+ };
1576
+ /** The ImportWarning codes the platform can attach to an import. */
1577
+ import_warning_codes: string[];
1578
+ /** The closed error-code union, enumerated at runtime. */
1579
+ error_codes: string[];
1580
+ }
1581
+
1582
+ export { type AgentResult as $, type AgentsClient as A, type JobCreate as B, type CapabilityDocument as C, type DatasetsClient as D, EVAL_SANDBOX_PROVIDERS as E, type Job as F, type JobFailure as G, type HostedClientConfig as H, type JobStats as I, type JobsClient as J, type AgentDatasetStats as K, type ListImportsOptions as L, type ManagedProviderCapability as M, type JobStatus as N, type EvalSandboxProvider as O, type ProviderCapability as P, type Trial as Q, type TrialArtifactStream as R, type StatusVocabulary as S, type TrialsClient as T, type UpstreamStatus as U, type TrialStatus as V, type TrialCounts as W, type TrialStatusTally as X, type TimingInfo as Y, type ModelInfo as Z, type AgentInfo as _, type AuthClient as a, type VerifierResult as a0, type ExceptionInfo as a1, type StepResult as a2, type AttemptPhase as a3, type StopResponse as a4, type TraceEvent as a5, type TraceEventPage as a6, type JobEvent as a7, type JobWatch as a8, type VerifierEnvironmentMode as a9, type TrialList as aA, type DatasetPage as aB, type DatasetList as aC, type AgentPage as aD, type AgentList as aE, type StartJobOptions as aF, type ListJobsOptions as aG, type ListTrialsOptions as aH, type ListDatasetsOptions as aI, type ListAgentsOptions as aJ, type GetDatasetOptions as aK, type TraceOptions as aL, type WatchJobOptions as aM, type WatchImportOptions as aN, type DownloadJobOptions as aO, type DownloadDatasetOptions as aP, type CompareJobAggregate as aa, type CompareCell as ab, type CompareCoverage as ac, type CompareTaskRow as ad, type CompareResponse as ae, type SourceJob as af, type ResumeRequest as ag, type RegradeRequest as ah, type DatasetImport as ai, type DatasetPatch as aj, type PublishDatasetInput as ak, type DatasetSource as al, type DatasetImportStatus as am, type DatasetImportFailure as an, type ImportWarning as ao, type Agent as ap, type AgentInput as aq, type AgentUpsertInput as ar, type AgentSource as as, type AgentSourceInput as at, type SpendSource as au, type Page as av, type PageOptions as aw, type JobPage as ax, type JobList as ay, type TrialPage as az, type HostedErrorCode as b, HOSTED_ERROR_CODES as c, TRIAL_ARTIFACT_STREAMS as d, TRIAL_STATUSES as e, type AgentCapability as f, type Awaitable as g, type DatasetImportList as h, isHostedErrorCode as i, type DatasetImportPage as j, type AuthStatus as k, type ApiKey as l, type JobTaskRollup as m, type JobTaskRollupPage as n, type JobTaskRollupList as o, type ListJobTasksOptions as p, type Dataset as q, type ActiveDataset as r, type DatasetVersion as s, type DatasetVersionState as t, type DatasetRef as u, type DatasetSelector as v, type Task as w, type TaskProviderVerdict as x, type AgentArm as y, type AgentArmInput as z };