runbios-sdk 0.2.1-rc.98 → 0.2.2-dev.171

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/dist/types.d.ts CHANGED
@@ -8,13 +8,13 @@ export interface BiOSConfig {
8
8
  orgId?: string;
9
9
  /** Workspace ID. Optional when using API keys (resolved from the key). Can override for multi-workspace keys. */
10
10
  workspaceId?: string;
11
- /** Base URL for the API. Falls back to the RUNBIOS_BASE_URL environment variable (legacy: BIOS_BASE_URL), then the canonical https://api-staging.runbios.ai hostname. */
11
+ /** Base URL for the API. Falls back to the RUNBIOS_BASE_URL environment variable (legacy: BIOS_BASE_URL), then the canonical https://api-dev.runbios.ai hostname. */
12
12
  baseUrl?: string;
13
13
  /** Request timeout in milliseconds. Defaults to 30000. */
14
14
  timeout?: number;
15
15
  /** Default per-deployment inference key. Can be overridden per inference call. */
16
16
  inferenceKey?: string;
17
- /** Inference base URL. Defaults to baseUrl, then https://api-staging.runbios.ai. */
17
+ /** Inference base URL. Defaults to baseUrl, then https://api-dev.runbios.ai. */
18
18
  inferenceBaseUrl?: string;
19
19
  /** End-to-end inference timeout in milliseconds. Defaults to 15 minutes. */
20
20
  inferenceTimeout?: number;
@@ -396,8 +396,9 @@ export interface DatasetPreview {
396
396
  }
397
397
  /** Parameters for importing a dataset from HuggingFace Hub. */
398
398
  export interface DatasetImportHFParams {
399
- /** HuggingFace dataset repository ID (e.g. "databricks/dolly-15k"). */
399
+ /** HuggingFace dataset repository ID (e.g. "HuggingFaceH4/ultrachat_200k"). */
400
400
  repoId: string;
401
+ revision?: string;
401
402
  /** Display name for the imported dataset. */
402
403
  name?: string;
403
404
  /** Dataset subset/config to import. */
@@ -418,6 +419,7 @@ export interface DatasetImportHFParams {
418
419
  /** Parameters for registering a HuggingFace dataset without an integration. */
419
420
  export interface DatasetRegisterHFParams {
420
421
  repoId: string;
422
+ revision?: string;
421
423
  name?: string;
422
424
  workspaceId?: string;
423
425
  description?: string;
@@ -441,6 +443,7 @@ export interface DatasetHubSearchParams {
441
443
  export interface DatasetHubPreviewParams {
442
444
  /** HuggingFace dataset ID. */
443
445
  datasetId: string;
446
+ integrationId?: string;
444
447
  /** Split to preview. Defaults to "train". */
445
448
  split?: string;
446
449
  /** Subset/config name. */
@@ -545,6 +548,21 @@ export interface TrainingCreateParams {
545
548
  * large dataset without importing a trimmed copy.
546
549
  */
547
550
  datasetSampleLimits?: Record<string, number>;
551
+ /**
552
+ * Per-dataset sampling (key = a dataset ID from datasetIds): exactly one of
553
+ * `rows` or `percent` (of the dataset's usable rows for this method). Asking
554
+ * for more than the dataset holds uses every usable row and the run says
555
+ * so. Resolved to a row count identically by preflight and create, recorded
556
+ * in the mix manifest, and reproduced exactly on resume.
557
+ */
558
+ datasetSampling?: Record<string, DatasetSampling>;
559
+ /**
560
+ * How sampled rows are chosen: `first` (default) keeps the first N usable
561
+ * rows in file order -- what a curriculum wants; `random` draws a seeded
562
+ * uniform sample of the usable rows (kept in file order, seed = the mixing
563
+ * plan's seed, so a resume draws the same rows).
564
+ */
565
+ datasetSamplingStrategy?: 'first' | 'random';
548
566
  /** Training method. */
549
567
  method: TrainingMethod;
550
568
  /** Adapter type. */
@@ -565,8 +583,10 @@ export interface TrainingCreateParams {
565
583
  name?: string;
566
584
  /** Target workspace ID. */
567
585
  workspaceId?: string;
568
- /** Number of training epochs. */
586
+ /** Number of training epochs. Governs the full run when maxSteps is omitted. */
569
587
  epochs?: number;
588
+ /** Optional hard step cap that overrides epochs. Omit for no cap and all configured epochs. */
589
+ maxSteps?: number;
570
590
  /** Training batch size per device. */
571
591
  batchSize?: number;
572
592
  /** Gradient accumulation steps. */
@@ -609,7 +629,7 @@ export interface TrainingCreateParams {
609
629
  integrationId?: string;
610
630
  /** Existing network volume to attach. */
611
631
  networkVolumeId?: string;
612
- /** Cache the composed dataset for retry/resume. Defaults to true server-side. */
632
+ /** @deprecated No effect. Prepared data is transient; exact resume rebuilds it from pinned source metadata and verifies its checksum. */
613
633
  cacheDataset?: boolean;
614
634
  /** Legacy dataset ordering mode. Prefer mixing for weighted/phased plans. */
615
635
  datasetMixing?: 'shuffle' | 'sequential' | 'interleave' | 'random' | 'curriculum';
@@ -681,6 +701,13 @@ export interface TrainingJob {
681
701
  status: TrainingJobStatus;
682
702
  error_message?: string | null;
683
703
  error_code?: string | null;
704
+ /**
705
+ * Who ended a `stopped` run: `user` (Stop button, API, queue cancel) or
706
+ * `platform` (wallet exhausted, account block). Empty when the run is not
707
+ * stopped, or was stopped before this was recorded. A `failed` run is a
708
+ * different fact and never carries a stop origin.
709
+ */
710
+ stop_origin?: 'user' | 'platform' | '';
684
711
  current_step?: number;
685
712
  total_steps?: number;
686
713
  current_loss?: number | null;
@@ -754,12 +781,37 @@ export interface TrainingListResponse {
754
781
  limit: number;
755
782
  offset: number;
756
783
  }
784
+ export interface TrainingDeviceMetrics {
785
+ memory_used_mib: number | null;
786
+ memory_total_mib: number | null;
787
+ utilization_pct: number | null;
788
+ temperature_c: number | null;
789
+ }
790
+ export interface TrainingResourceMetrics {
791
+ scope: 'worker';
792
+ sampled_at: string | null;
793
+ received_at: string;
794
+ gpu_count: number | null;
795
+ gpu_memory_used_mib: number | null;
796
+ gpu_memory_total_mib: number | null;
797
+ gpu_utilization_pct: number | null;
798
+ gpu_temperature_c: number | null;
799
+ host_ram_used_gib: number | null;
800
+ host_ram_limit_gib: number | null;
801
+ host_ram_pct: number | null;
802
+ host_cpu_usage_pct: number | null;
803
+ host_cpu_limit_cores: number | null;
804
+ disk_used_gib: number | null;
805
+ disk_total_gib: number | null;
806
+ gpus: TrainingDeviceMetrics[];
807
+ }
757
808
  /** Training metrics for a job. */
758
809
  export interface TrainingMetrics {
759
810
  metrics: MetricPoint[];
760
811
  training_method?: string | null;
761
812
  rlhf_type?: string | null;
762
813
  graph_configs: MetricGraphConfig[];
814
+ resource_metrics?: TrainingResourceMetrics | null;
763
815
  /** @deprecated Use metrics. Populated as a compatibility alias. */
764
816
  steps?: MetricPoint[];
765
817
  }
@@ -877,6 +929,8 @@ export interface TrainingPreflightDataset {
877
929
  export interface TrainingPreflightWarning {
878
930
  code: string;
879
931
  message: string;
932
+ /** Request field the warning is about (e.g. `per_device_train_batch_size`), when there is one. */
933
+ field?: string;
880
934
  }
881
935
  /** Side-effect-free validation/sizing result; this endpoint never creates or bills a job. */
882
936
  export interface TrainingPreflightResponse {
@@ -897,6 +951,131 @@ export interface TrainingPreflightResponse {
897
951
  queue_eligible: boolean;
898
952
  warnings: TrainingPreflightWarning[];
899
953
  checked_at: string;
954
+ /**
955
+ * The trainer image's own sizing verdict for the requested GPU shape:
956
+ * recommended microbatch/accumulation/learning rate, predicted peak memory,
957
+ * minimum GPU count, wall-clock estimate, and whether the per-device batch
958
+ * you asked for is predicted to fit. Absent when no advisor is deployed.
959
+ */
960
+ advisor?: TrainingAdvisorVerdict;
961
+ }
962
+ /** Parameters for `training.recommend()` (GET /api/training/recommend). */
963
+ export interface TrainingRecommendParams {
964
+ model: string;
965
+ modelRevision?: string;
966
+ integrationId?: string;
967
+ gpuType: string;
968
+ gpuCount?: number;
969
+ adapter?: 'full' | 'lora' | 'qlora';
970
+ method?: 'sft' | 'cpt' | 'pt';
971
+ maxLength?: number;
972
+ epochs?: number;
973
+ maxSteps?: number;
974
+ /** Your own microbatch, to be judged against the model. */
975
+ perDeviceTrainBatchSize?: number;
976
+ gradientAccumulationSteps?: number;
977
+ /** Datasets the job will train on; their measured token statistics feed the sizing. */
978
+ datasetIds?: string[];
979
+ workspaceId?: string;
980
+ }
981
+ /** Basis of one recommended value: measured on hardware, derived through a stated model, or an argued default. */
982
+ export type TrainingAdvisorBasis = 'measured' | 'derived' | 'judgement';
983
+ export interface TrainingAdvisorJustification {
984
+ field: string;
985
+ value: string;
986
+ reason: string;
987
+ basis: TrainingAdvisorBasis;
988
+ }
989
+ export interface TrainingAdvisorRecommendation {
990
+ per_device_train_batch_size: number;
991
+ gradient_accumulation_steps: number;
992
+ global_batch_size: number;
993
+ activation_checkpoint: string;
994
+ compile: boolean;
995
+ learning_rate: number;
996
+ warmup_steps: number;
997
+ parallelism: {
998
+ dp_replicate: number;
999
+ dp_shard: number;
1000
+ tp: number;
1001
+ pp: number;
1002
+ cp: number;
1003
+ ep: number;
1004
+ };
1005
+ /** Predicted peak reserved GPU memory per device, GiB. */
1006
+ predicted_peak_gb: number;
1007
+ /** (median, worst) percent the prediction ran over measurement on the calibration rows. */
1008
+ memory_band_percent: [number, number];
1009
+ memory_class: string;
1010
+ predicted_mfu: number;
1011
+ supervised_tokens_per_step: number;
1012
+ predicted_roughness: number;
1013
+ tokens_per_second: number;
1014
+ throughput_basis: 'measured' | 'derived';
1015
+ wall_clock: {
1016
+ steps_per_epoch: number;
1017
+ total_steps: number;
1018
+ training_hours: number;
1019
+ startup_minutes_estimate: number;
1020
+ basis: TrainingAdvisorBasis;
1021
+ } | null;
1022
+ }
1023
+ export interface TrainingAdvisorUserShape {
1024
+ per_device_train_batch_size: number;
1025
+ fits: boolean;
1026
+ largest_fitting_batch?: number;
1027
+ predicted_peak_gb?: number;
1028
+ reason?: string;
1029
+ /** Same global batch, a microbatch that fits. Present only when `fits` is false. */
1030
+ suggested?: {
1031
+ per_device_train_batch_size: number;
1032
+ gradient_accumulation_steps: number;
1033
+ reason: string;
1034
+ };
1035
+ }
1036
+ /**
1037
+ * Verdict of the training advisor. `available: false` means no advisor is
1038
+ * deployed or it did not answer; nothing else is populated then.
1039
+ */
1040
+ export interface TrainingAdvisorVerdict {
1041
+ available: boolean;
1042
+ reason?: string;
1043
+ fits?: boolean;
1044
+ /** Smallest GPU count of this type the job fits on; null when none up to the per-job cap. */
1045
+ min_gpu_count?: number | null;
1046
+ gpu?: {
1047
+ platform_type: string;
1048
+ sized_as: string;
1049
+ capacity_gb: number;
1050
+ capacity_measured: boolean;
1051
+ };
1052
+ model?: {
1053
+ params_total_b: number;
1054
+ params_active_b: number;
1055
+ is_moe: boolean;
1056
+ is_vlm: boolean;
1057
+ has_linear_attention: boolean;
1058
+ };
1059
+ dataset_measured?: boolean;
1060
+ seq_len?: number;
1061
+ gpu_count?: number;
1062
+ recommended?: TrainingAdvisorRecommendation;
1063
+ user_shape?: TrainingAdvisorUserShape;
1064
+ justifications?: TrainingAdvisorJustification[];
1065
+ warnings?: string[];
1066
+ dataset_stats_used?: {
1067
+ num_rows: number;
1068
+ avg_tokens_per_sample?: number;
1069
+ avg_supervised_tokens_per_sample?: number;
1070
+ has_images: boolean;
1071
+ estimated: boolean;
1072
+ };
1073
+ measured_history?: {
1074
+ tokens_per_second: number;
1075
+ memory_anchors: number;
1076
+ };
1077
+ model_id?: string;
1078
+ model_revision?: string;
900
1079
  }
901
1080
  /** One method, algorithm, or adapter reported by the pinned training engine. */
902
1081
  export interface TrainingCapabilityChoice {
@@ -916,6 +1095,8 @@ export interface TrainingConfigFieldCapability {
916
1095
  label: string;
917
1096
  type: 'integer' | 'number' | 'boolean' | 'string' | 'string_array' | 'string_or_string_array' | 'object';
918
1097
  default: unknown;
1098
+ default_by_adapter?: Record<string, number>;
1099
+ requires_step_horizon?: boolean;
919
1100
  enabled: boolean;
920
1101
  disabled_reason?: string;
921
1102
  aliases?: string[];
@@ -1052,6 +1233,12 @@ export interface GPUPricingResponse {
1052
1233
  stale?: boolean;
1053
1234
  stale_message?: string;
1054
1235
  }
1236
+ /** One dataset's sampling request: exactly one of rows or percent. */
1237
+ export interface DatasetSampling {
1238
+ rows?: number;
1239
+ /** Percent (0-100) of the dataset's usable rows for the training method. */
1240
+ percent?: number;
1241
+ }
1055
1242
  /** Parameters understood by the authenticated training GPU-options endpoint. */
1056
1243
  export interface GPUOptionsParams {
1057
1244
  modelId: string;
@@ -1067,6 +1254,17 @@ export interface GPUOptionsParams {
1067
1254
  rlhfType?: RLHFAlgorithm | string;
1068
1255
  modelParamsB?: number;
1069
1256
  modelActiveParamsB?: number;
1257
+ /** Training sequence length the sizing should assume (tokens). */
1258
+ maxLength?: number;
1259
+ /**
1260
+ * Effective batch (samples per optimizer step) -- the quality decision made
1261
+ * before a GPU is chosen. Each option then answers with the per-device split
1262
+ * that runs it there (`recommended_micro_batch` x `recommended_grad_accum`
1263
+ * x `required_count`), and the minimum GPU count is sized at micro-batch 1.
1264
+ */
1265
+ effectiveBatch?: number;
1266
+ /** Per-device batch to size as typed instead (ignored when effectiveBatch is set). */
1267
+ perDeviceTrainBatchSize?: number;
1070
1268
  }
1071
1269
  /** One model-aware GPU option with live stock and total-price context. */
1072
1270
  export interface GPUOption {
@@ -1087,6 +1285,10 @@ export interface GPUOption {
1087
1285
  bookable: boolean;
1088
1286
  reason?: string;
1089
1287
  checked_at?: string;
1288
+ /** Present when the request stated an effective batch: its split on this card at required_count. */
1289
+ effective_batch?: number;
1290
+ recommended_micro_batch?: number;
1291
+ recommended_grad_accum?: number;
1090
1292
  }
1091
1293
  /** Actionable alternative returned when no GPU option is currently bookable. */
1092
1294
  export interface GPUOptionSuggestion {
@@ -1116,6 +1318,10 @@ export interface GPUOptionsResponse {
1116
1318
  storage_gb: number;
1117
1319
  price_per_hour_cents: number;
1118
1320
  total_price_per_hour_cents: number;
1321
+ /** Present when the request stated an effective batch. */
1322
+ effective_batch?: number;
1323
+ per_device_train_batch_size?: number;
1324
+ gradient_accumulation_steps?: number;
1119
1325
  };
1120
1326
  suggestions?: GPUOptionSuggestion[];
1121
1327
  }
@@ -1817,7 +2023,7 @@ export interface ApiKey {
1817
2023
  * on purpose: a scope added to the catalog must not make an existing SDK build
1818
2024
  * reject a key it just read back from the API.
1819
2025
  */
1820
- export type ApiKeyScope = 'models:read' | 'datasets:read' | 'datasets:write' | 'training:read' | 'training:write' | 'billing:read' | 'deployments:read' | 'deployments:write' | 'analytics:read' | 'integrations:read' | 'integrations:write' | 'skills:read' | 'skills:write' | 'serverless' | (string & {});
2026
+ export type ApiKeyScope = 'models:read' | 'datasets:read' | 'datasets:write' | 'training:read' | 'training:write' | 'billing:read' | 'deployments:read' | 'deployments:write' | 'analytics:read' | 'integrations:read' | 'integrations:write' | 'skills:read' | 'skills:write' | 'loop:read' | 'loop:write' | 'serverless' | (string & {});
1821
2027
  /** Result of introspecting an API key — shows what it can do. */
1822
2028
  export interface ApiKeyIntrospection {
1823
2029
  auth_type: 'api_key' | 'jwt';
@@ -1953,3 +2159,1771 @@ export interface ChatCompletionUsage {
1953
2159
  total_tokens: number;
1954
2160
  prompt_tokens_details?: PromptTokensDetails;
1955
2161
  }
2162
+ /** What a training set is shaped for. The three need different things. */
2163
+ export type LoopMethod = 'sft' | 'dpo' | 'grpo' | 'kto';
2164
+ /** Who judged an answer. */
2165
+ export type LoopSource = 'human' | 'verifier' | 'judge' | 'behavioural';
2166
+ /** The judgement itself. */
2167
+ /**
2168
+ * The judgement recorded on an answer.
2169
+ *
2170
+ * `accepted` / `rejected` are the thumbs (see `upvote` / `downvote`).
2171
+ * `edited` says the model was WRONG and carries the better answer.
2172
+ * `gold` records the reference answer for the question, making no claim about
2173
+ * whether the model was right — which is why it is separate from `edited`.
2174
+ */
2175
+ export type LoopVerdict = 'accepted' | 'rejected' | 'edited' | 'scored' | 'gold';
2176
+ /** A function an assistant turn asked to invoke. `arguments` is JSON-encoded. */
2177
+ export interface LoopToolCall {
2178
+ id?: string;
2179
+ type?: string;
2180
+ function: {
2181
+ name: string;
2182
+ arguments: string;
2183
+ };
2184
+ }
2185
+ /** One conversational turn, in the chat-completions shape. */
2186
+ export interface LoopMessage {
2187
+ role: string;
2188
+ content?: string;
2189
+ tool_calls?: LoopToolCall[];
2190
+ tool_call_id?: string;
2191
+ name?: string;
2192
+ }
2193
+ export interface LoopCaptureParams {
2194
+ /** The source being recorded. Must be enabled first, or nothing is stored. */
2195
+ deployment_id: string;
2196
+ model: string;
2197
+ messages: LoopMessage[];
2198
+ completion?: string;
2199
+ /** What the answering turn invoked, if anything. */
2200
+ tool_calls?: LoopToolCall[];
2201
+ /** The tool schema the model was offered. Without it, tool training invents names. */
2202
+ tools?: unknown;
2203
+ model_version?: string;
2204
+ /** Ties multi-turn work together. */
2205
+ conversation_id?: string;
2206
+ /** Your own idempotency handle. Replaying it returns the same trace. */
2207
+ request_id?: string;
2208
+ prompt_tokens?: number;
2209
+ completion_tokens?: number;
2210
+ latency_ms?: number;
2211
+ metadata?: Record<string, unknown>;
2212
+ }
2213
+ export interface LoopCaptureResult {
2214
+ captured: boolean;
2215
+ /** Present when captured. Ours, never the id you sent. */
2216
+ trace_id?: string;
2217
+ /** Set when something was removed before the record was written. */
2218
+ redacted?: boolean;
2219
+ download_url?: string;
2220
+ /** Why nothing was stored, e.g. `capture_disabled`, `not_sampled`. */
2221
+ reason?: string;
2222
+ }
2223
+ export interface LoopImportParams {
2224
+ /**
2225
+ * Where this came from. Required, and becomes the id every imported
2226
+ * conversation is filed under, so the import stays sliceable later.
2227
+ */
2228
+ source: string;
2229
+ /** The file, already parsed. At most 5000 rows per call. */
2230
+ rows: Record<string, unknown>[];
2231
+ /** Which model produced these, when the rows do not say per-row. */
2232
+ model?: string;
2233
+ /** Applied to every row: importing a dump is when somebody knows what it is. */
2234
+ labels?: string[];
2235
+ attributes?: Record<string, string>;
2236
+ /**
2237
+ * Repeat the `import_id` a previous call reported to make this call a RETRY
2238
+ * of that one: rows that already arrived come back under `already_present`
2239
+ * instead of being stored again.
2240
+ *
2241
+ * Leave it out and this call is its own import. The endpoint will NOT guess:
2242
+ * two calls carrying the same rows are as likely to be two pages of one
2243
+ * export as one call sent twice, and guessing "retry" silently drops the
2244
+ * second copy of every conversation a file lists more than once. Every
2245
+ * response carries the token it was filed under, so a retry is always
2246
+ * available and never has to be inferred.
2247
+ *
2248
+ * SEND THE ROWS AS YOU SENT THEM. Without `row_ids` a row is matched on its
2249
+ * text, and the match survives re-ordered keys and different whitespace but
2250
+ * NOT a number that has been re-spelled: `100`, `1e2` and `100.0` are three
2251
+ * different rows. A round trip through most JSON libraries re-spells numbers
2252
+ * (Python turns `1e2` into `100.0`), so a retry built by re-serialising a
2253
+ * parsed file can store rows a second time. Keep the bytes you sent, or send
2254
+ * `row_ids` and stop depending on the text at all.
2255
+ */
2256
+ import_id?: string;
2257
+ /**
2258
+ * Where this page starts in the file, counting from 0. Send it when you split
2259
+ * one import across several calls under one `import_id`, so two pages are not
2260
+ * matched against each other row for row. Ignored when you send `row_ids`.
2261
+ */
2262
+ row_offset?: number;
2263
+ /**
2264
+ * Your own id for each row, in the same order as `rows`: a ticket number, a
2265
+ * conversation id, whatever the export already carries.
2266
+ *
2267
+ * The strongest form of identity, and the one to reach for with a file big
2268
+ * enough to page. It SUPERSEDES `import_id` and `row_offset`: with it,
2269
+ * chunking, ordering and subsets all stop mattering and any part of the file
2270
+ * can be re-sent exactly. One per row or none at all, and no two rows in a
2271
+ * call may share an id: both are refused rather than silently merging two
2272
+ * conversations into one.
2273
+ *
2274
+ * THE ID IS THE IDENTITY, AND THE ROW'S TEXT IS NOT PART OF IT. A row sent
2275
+ * again under an id this source already imported comes back under
2276
+ * `already_present` and the stored conversation is left as it was, even if
2277
+ * you changed its text. A CORRECTION DOES NOT LAND THIS WAY: send it under a
2278
+ * new id. That is the trade for making every retry, page and subset safe -
2279
+ * the text is never compared, so nothing your serialiser does to it can turn
2280
+ * a retry into a second import.
2281
+ */
2282
+ row_ids?: string[];
2283
+ }
2284
+ /**
2285
+ * What an import did.
2286
+ *
2287
+ * A call where every row stored resolves with this. A call where SOME rows
2288
+ * could not be stored rejects with {@link LoopImportIncompleteError}, whose
2289
+ * `outcome` is this same shape: the counts are the whole answer either way, so
2290
+ * read them from the error exactly as you would from a success.
2291
+ */
2292
+ export interface LoopImportResult {
2293
+ /**
2294
+ * The token this call was filed under, whether you sent one or it was minted
2295
+ * for you. Send the same rows again with this `import_id` to retry the call.
2296
+ * Present on every response, success or failure, because a call can succeed
2297
+ * and still need retrying when the answer never reached you.
2298
+ */
2299
+ import_id?: string;
2300
+ /** How many conversations this call created. */
2301
+ imported: number;
2302
+ /**
2303
+ * How many rows an earlier call under the SAME `import_id` (or carrying the
2304
+ * same `row_ids`) had already imported, and so were not stored a second time.
2305
+ * Zero on a first import, and zero on any call that repeated neither: a call
2306
+ * that does not say it is a retry is its own import, which is what lets a
2307
+ * file sent in pages land complete.
2308
+ */
2309
+ already_present?: number;
2310
+ /**
2311
+ * How many verdicts THIS CALL WROTE. Reported apart from `imported` because
2312
+ * it is the difference between data you can train on and data somebody still
2313
+ * has to look at.
2314
+ *
2315
+ * Re-sending a row whose verdict already landed writes nothing and counts
2316
+ * nothing, so a plain retry reads 0. Re-sending one that came back in
2317
+ * `verdicts_not_saved` DOES count here when the write succeeds this time,
2318
+ * even though the row itself is counted under `already_present`: that is the
2319
+ * first time anybody recorded that verdict, not a second reviewer. This is
2320
+ * the field to read to confirm a repair landed.
2321
+ */
2322
+ reviewed: number;
2323
+ needs_review: number;
2324
+ /**
2325
+ * How many rows could not be READ as a conversation. The file's shape is
2326
+ * wrong, and changing the file is what fixes it.
2327
+ */
2328
+ refused: number;
2329
+ /**
2330
+ * How many rows were understood and then could not be stored. Counted apart
2331
+ * from `refused`, because the two are different claims and rewriting a file
2332
+ * that was already correct fixes nothing. A call with any of these rejects
2333
+ * rather than resolving; read it off {@link LoopImportIncompleteError.outcome}.
2334
+ */
2335
+ not_saved?: number;
2336
+ /**
2337
+ * WHICH rows those were, counting from 1 in the order you sent them, so you
2338
+ * can say what is missing rather than only how much.
2339
+ *
2340
+ * NOT A SMALLER FILE TO SEND BACK unless you sent `row_ids`. Without them a
2341
+ * row is identified by its place in the call, so a shorter list moves every
2342
+ * row after the gap and each one is stored again. Re-send the whole set of
2343
+ * rows, in the same order, with the same `import_id`; what already arrived
2344
+ * comes back under `already_present`. With `row_ids` a row carries its own
2345
+ * identity and any subset is exact.
2346
+ */
2347
+ not_saved_rows?: number[];
2348
+ /**
2349
+ * Rows whose CONVERSATION stored and whose VERDICT did not, and which rows
2350
+ * those were. A preference pair or a thumbs label is two writes, and the
2351
+ * second can fail on its own: the conversation is then in the loop carrying
2352
+ * nobody's judgement, counted under `needs_review` rather than `reviewed`,
2353
+ * and not trainable. The call rejects with {@link LoopImportIncompleteError},
2354
+ * and sending the rows again under the same `import_id` records the verdict
2355
+ * without storing the conversation twice.
2356
+ */
2357
+ verdicts_not_saved?: number;
2358
+ verdicts_not_saved_rows?: number[];
2359
+ /**
2360
+ * How many conversations of each shape THIS CALL created. Rows that were
2361
+ * refused, rows that failed to store and rows an earlier import already had
2362
+ * are not in here: the counts and the sentences in `notes` describe the same
2363
+ * call, so neither can claim a conversation is waiting for review when none
2364
+ * was written.
2365
+ */
2366
+ by_shape: Record<string, number>;
2367
+ refused_why: Record<string, number>;
2368
+ /**
2369
+ * The conversations from this file that are now in the loop, in file order:
2370
+ * the ones this call created and the ones it found already there. Without
2371
+ * them a caller whose import half-failed has a number and no way to reach
2372
+ * what landed: no way to label it, review it, or delete it and start again.
2373
+ */
2374
+ trace_ids?: string[];
2375
+ notes: string[];
2376
+ }
2377
+ export interface LoopSignal {
2378
+ id: string;
2379
+ trace_id: string;
2380
+ source: LoopSource;
2381
+ verdict: LoopVerdict;
2382
+ correction: string | null;
2383
+ score: number | null;
2384
+ ground_truth: string | null;
2385
+ reason: string | null;
2386
+ author: string | null;
2387
+ created_at: string;
2388
+ }
2389
+ export interface LoopSignalParams {
2390
+ verdict: LoopVerdict;
2391
+ source?: LoopSource;
2392
+ /** Required when verdict is `edited`. The answer the model should have given. */
2393
+ correction?: string;
2394
+ /** Required when verdict is `scored`. */
2395
+ score?: number;
2396
+ /** A value or fact the answer can be checked against. GRPO needs one. */
2397
+ ground_truth?: string;
2398
+ reason?: string;
2399
+ author?: string;
2400
+ metadata?: Record<string, unknown>;
2401
+ }
2402
+ export interface LoopTrace {
2403
+ id: string;
2404
+ workspace_id: string;
2405
+ deployment_id: string | null;
2406
+ model: string;
2407
+ model_version: string | null;
2408
+ messages: LoopMessage[];
2409
+ completion: string;
2410
+ tool_calls?: LoopToolCall[];
2411
+ tools?: unknown;
2412
+ conversation_id: string | null;
2413
+ request_id: string | null;
2414
+ prompt_tokens: number | null;
2415
+ completion_tokens: number | null;
2416
+ latency_ms: number | null;
2417
+ redacted_at: string | null;
2418
+ /** What the redaction pass removed, counted by rule. */
2419
+ redaction_report?: Record<string, number>;
2420
+ created_at: string;
2421
+ expires_at: string;
2422
+ signals?: LoopSignal[];
2423
+ }
2424
+ export interface LoopTraceListParams {
2425
+ deployment_id?: string;
2426
+ conversation_id?: string;
2427
+ from?: string;
2428
+ to?: string;
2429
+ /** Only conversations that already carry a verdict. */
2430
+ signalled?: boolean;
2431
+ /** One bare tag. A tag with children matches them too. */
2432
+ label?: string;
2433
+ /**
2434
+ * Named dimensions, ANDed together: `{ category: 'billing', language: 'es' }`.
2435
+ * Exact within their key — `source=support` does not match `team=support`.
2436
+ */
2437
+ attributes?: Record<string, string>;
2438
+ /** Only the conversations nobody has described yet. */
2439
+ unlabelled?: boolean;
2440
+ limit?: number;
2441
+ offset?: number;
2442
+ /** Only what ONE model answered — the filter distillation is made of. */
2443
+ model?: string;
2444
+ /** `captured` (our model's own behaviour) or `imported` (a file somebody brought). */
2445
+ origin?: 'captured' | 'imported';
2446
+ }
2447
+ export interface LoopTraceListResponse {
2448
+ traces: LoopTrace[];
2449
+ total: number;
2450
+ limit: number;
2451
+ offset: number;
2452
+ }
2453
+ export interface LoopDatasetCreateParams {
2454
+ name: string;
2455
+ method: LoopMethod;
2456
+ deployment_id?: string;
2457
+ from?: string;
2458
+ to?: string;
2459
+ /** Narrow to one slice. A label with children selects them too. */
2460
+ label?: string;
2461
+ /**
2462
+ * Take this many of what the filters matched. Which ones is arbitrary but
2463
+ * REPEATABLE: the same conversations always give the same slice, so two
2464
+ * builds of one spec describe the same set.
2465
+ */
2466
+ sample?: number;
2467
+ /** Holds back your most recent work, not a random slice. 0-50. */
2468
+ holdout_percent?: number;
2469
+ max_items?: number;
2470
+ /**
2471
+ * Named dimensions, ANDed with each other and with `label`:
2472
+ * `{ category: 'billing', language: 'es' }`. This is what turns one captured
2473
+ * corpus into a different dataset for every task somebody trains for.
2474
+ */
2475
+ attributes?: Record<string, string>;
2476
+ /** Only the conversations nobody has described yet. */
2477
+ unlabelled?: boolean;
2478
+ /** Only what ONE model answered — the filter distillation is made of. */
2479
+ model?: string;
2480
+ /** `captured` (our model's own behaviour) or `imported` (a file somebody brought). */
2481
+ origin?: 'captured' | 'imported';
2482
+ }
2483
+ export interface LoopDataset {
2484
+ id: string;
2485
+ workspace_id: string;
2486
+ name: string;
2487
+ method: LoopMethod;
2488
+ status: string;
2489
+ spec: Record<string, unknown>;
2490
+ item_count: number;
2491
+ considered_count: number;
2492
+ /** Why rows were left out, by reason. */
2493
+ rejected_counts: Record<string, number>;
2494
+ holdout_count: number;
2495
+ holdout_cutoff: string | null;
2496
+ created_by: string | null;
2497
+ created_at: string;
2498
+ completed_at: string | null;
2499
+ download_url: string;
2500
+ }
2501
+ export interface LoopDatasetListParams {
2502
+ method?: LoopMethod;
2503
+ limit?: number;
2504
+ offset?: number;
2505
+ }
2506
+ export interface LoopDatasetItem {
2507
+ id: number;
2508
+ trace_id: string;
2509
+ split: 'train' | 'holdout';
2510
+ payload: Record<string, unknown>;
2511
+ dedup_key: string;
2512
+ created_at: string;
2513
+ /** The conversation this row was built from. */
2514
+ trace_url: string;
2515
+ }
2516
+ export interface LoopConfig {
2517
+ deployment_id: string;
2518
+ workspace_id: string;
2519
+ enabled: boolean;
2520
+ retention_days: number;
2521
+ sample_rate: number;
2522
+ enabled_by: string | null;
2523
+ enabled_at: string | null;
2524
+ updated_at: string;
2525
+ auto_grade: boolean;
2526
+ }
2527
+ export interface LoopBuildRuleParams {
2528
+ name: string;
2529
+ method: string;
2530
+ /**
2531
+ * How many rows reviewed SINCE THE LAST BUILD must exist before this fires
2532
+ * again. Minimum 10. Counting the whole corpus instead would fire the rule
2533
+ * every interval forever, because a total that has crossed a threshold stays
2534
+ * across it.
2535
+ */
2536
+ min_new_rows?: number;
2537
+ enabled?: boolean;
2538
+ /** The same selection `createDataset` takes, replayed verbatim. */
2539
+ spec?: LoopDatasetCreateParams;
2540
+ }
2541
+ export interface LoopBuildRule {
2542
+ id: string;
2543
+ workspace_id: string;
2544
+ name: string;
2545
+ method: string;
2546
+ spec: Record<string, unknown>;
2547
+ min_new_rows: number;
2548
+ enabled: boolean;
2549
+ last_built_at: string | null;
2550
+ last_dataset_id: string | null;
2551
+ /**
2552
+ * Why the rule did not fire last time it was checked. A rule quiet because
2553
+ * it is waiting looks exactly like one quiet because it is broken; this is
2554
+ * the difference.
2555
+ */
2556
+ last_reason: string | null;
2557
+ last_checked_at: string | null;
2558
+ created_by: string | null;
2559
+ created_at: string;
2560
+ updated_at: string;
2561
+ }
2562
+ export interface LoopConfigParams {
2563
+ enabled: boolean;
2564
+ retention_days?: number;
2565
+ /** Deterministic per conversation, so turns are never split. 0 < rate <= 1. */
2566
+ sample_rate?: number;
2567
+ /**
2568
+ * Continuous rule-based scoring for this source. OMITTED MEANS UNCHANGED:
2569
+ * a caller saving retention must not switch grading off for a workspace
2570
+ * that turned it on.
2571
+ */
2572
+ auto_grade?: boolean;
2573
+ }
2574
+ /** One thing a judge scores, separately from the others. */
2575
+ export interface LoopJudgeDimension {
2576
+ key: string;
2577
+ description?: string;
2578
+ }
2579
+ /** Which slice of the corpus a judge is responsible for. */
2580
+ export interface LoopJudgeSelection {
2581
+ label?: string;
2582
+ deployment_id?: string;
2583
+ from?: string;
2584
+ to?: string;
2585
+ sample?: number;
2586
+ /** Skip conversations that already carry a judge verdict. */
2587
+ only_unscored?: boolean;
2588
+ }
2589
+ export interface LoopJudge {
2590
+ id: string;
2591
+ workspace_id: string;
2592
+ name: string;
2593
+ instructions: string;
2594
+ dimensions: LoopJudgeDimension[];
2595
+ selection: LoopJudgeSelection;
2596
+ model: string | null;
2597
+ write_gold: boolean;
2598
+ enabled: boolean;
2599
+ /** The platform's agent runs this judge, on the workspace's own serverless account. */
2600
+ auto: boolean;
2601
+ /**
2602
+ * Was automatic when the agent was turned off. Turning the agent back on
2603
+ * restores exactly these judges.
2604
+ */
2605
+ auto_paused: boolean;
2606
+ /**
2607
+ * Why the last turn-on did NOT restore this judge. Null is the ordinary
2608
+ * state. Set when the resume declined to make it automatic because the model
2609
+ * it names is not one the serving gateway will route — handing it back would
2610
+ * buy a run that fails every conversation. It stays paused, so pointing it at
2611
+ * a model that routes and turning the agent on again brings it back.
2612
+ */
2613
+ auto_pause_reason: string | null;
2614
+ created_by: string | null;
2615
+ created_at: string;
2616
+ updated_at: string;
2617
+ }
2618
+ export interface LoopJudgeParams {
2619
+ name: string;
2620
+ instructions: string;
2621
+ dimensions: LoopJudgeDimension[];
2622
+ selection?: LoopJudgeSelection;
2623
+ model?: string;
2624
+ /**
2625
+ * Let this judge write the answer that should have been given, not just a
2626
+ * score. Off by default: a reference answer written by a model and then
2627
+ * trained on is distillation, which is a decision you make deliberately.
2628
+ */
2629
+ write_gold?: boolean;
2630
+ enabled?: boolean;
2631
+ /**
2632
+ * Hand the running of this judge to the platform's agent: new conversations
2633
+ * in its slice are scored as they arrive, one model call each, on THIS
2634
+ * WORKSPACE'S OWN serverless account (the agent spends through a managed key
2635
+ * of yours, "Conscious Loop" in your key list). Requires `model`. Off, you
2636
+ * run the model yourself with startRun / takeWork / postVerdicts.
2637
+ */
2638
+ auto?: boolean;
2639
+ }
2640
+ /** One pass of one rubric over one slice. */
2641
+ export interface LoopJudgeRun {
2642
+ id: string;
2643
+ judge_id: string;
2644
+ judge_name?: string;
2645
+ /**
2646
+ * `abandoned` is a run you opened and never drained: nothing was scored on
2647
+ * it for a week and items were still waiting. Nothing is deleted and posting
2648
+ * verdicts to it still works and still closes it as done. It exists so that
2649
+ * `open` keeps meaning "somebody is working on this".
2650
+ *
2651
+ * `stopped` is a run somebody closed on purpose before it finished, with
2652
+ * `stopRun`. Separate from `done` because a run that covered three of forty
2653
+ * conversations did not finish its work: every score it recorded is kept
2654
+ * either way, and reading one as the other overstates what was evaluated.
2655
+ */
2656
+ status: 'open' | 'done' | 'abandoned' | 'stopped';
2657
+ instructions: string;
2658
+ dimensions: LoopJudgeDimension[];
2659
+ model: string | null;
2660
+ selected: number;
2661
+ scored: number;
2662
+ failed: number;
2663
+ /** Who drains it: your own code ('caller') or the platform's agent ('platform'). */
2664
+ runner: 'caller' | 'platform';
2665
+ /** The agent's last complaint about this run, or null while it is working. */
2666
+ last_error: string | null;
2667
+ last_activity_at: string | null;
2668
+ created_by: string | null;
2669
+ created_at: string;
2670
+ finished_at: string | null;
2671
+ }
2672
+ /** The workspace's managed serverless key the agent spends through. Never the secret. */
2673
+ export interface LoopAgentCredential {
2674
+ workspace_id: string;
2675
+ key_id: string;
2676
+ key_prefix: string;
2677
+ created_by: string | null;
2678
+ created_at: string;
2679
+ updated_at: string;
2680
+ last_used_at: string | null;
2681
+ last_error: string | null;
2682
+ revoked_at: string | null;
2683
+ /**
2684
+ * The most the agent's key may spend on model calls in a calendar month,
2685
+ * in cents. `null` = no cap (the default): the agent can spend up to the
2686
+ * workspace's serverless balance. Once reached, the agent's model calls
2687
+ * are refused until next month and its runs pause with that reason.
2688
+ */
2689
+ monthly_spend_cap_cents: number | null;
2690
+ }
2691
+ export interface LoopAgentStatus {
2692
+ /** An agent can exist in this environment at all. */
2693
+ available: boolean;
2694
+ /** Which piece is missing when it cannot. */
2695
+ reason: string;
2696
+ /** A worker has checked in within the last minute. */
2697
+ online: boolean;
2698
+ agent: {
2699
+ seen_at: string;
2700
+ passes: number;
2701
+ judge_items: number;
2702
+ samples: number;
2703
+ } | null;
2704
+ credential: LoopAgentCredential | null;
2705
+ open_judge_runs: number;
2706
+ open_sample_runs: number;
2707
+ auto_judges: number;
2708
+ }
2709
+ export interface LoopSampleSelection extends LoopJudgeSelection {
2710
+ /** Skip conversations that already have alternatives. */
2711
+ only_unsampled?: boolean;
2712
+ }
2713
+ export interface LoopSampleRunParams {
2714
+ /** The model that writes the alternatives. For distillation, the teacher. */
2715
+ model: string;
2716
+ /** Alternatives per conversation, 1-8. Default 4. */
2717
+ n?: number;
2718
+ /** 0-2. Default 0.8; 0 makes every sample the same answer. */
2719
+ temperature?: number;
2720
+ /** 16-8192. Default 1024. */
2721
+ max_tokens?: number;
2722
+ /** Score each sample with this judge as it is written. */
2723
+ judge_id?: string;
2724
+ selection?: LoopSampleSelection;
2725
+ }
2726
+ /** "Write N alternatives to each conversation in this slice, and score them." */
2727
+ export interface LoopSampleRun {
2728
+ id: string;
2729
+ workspace_id: string;
2730
+ model: string;
2731
+ n: number;
2732
+ temperature: number;
2733
+ max_tokens: number;
2734
+ judge_id: string | null;
2735
+ judge_name: string | null;
2736
+ selection: LoopSampleSelection;
2737
+ status: 'open' | 'done';
2738
+ selected: number;
2739
+ done: number;
2740
+ failed: number;
2741
+ samples: number;
2742
+ last_error: string | null;
2743
+ last_activity_at: string | null;
2744
+ created_by: string | null;
2745
+ created_at: string;
2746
+ finished_at: string | null;
2747
+ }
2748
+ /**
2749
+ * One conversation to score, already rendered into the prompt to send.
2750
+ *
2751
+ * Send `prompt` as-is. Assembling it yourself is how two callers end up giving
2752
+ * the same rubric different instructions, and two judges given different
2753
+ * instructions are not one judge.
2754
+ */
2755
+ export interface LoopJudgeWorkItem {
2756
+ trace_id: string;
2757
+ messages: Array<{
2758
+ role: string;
2759
+ content?: string;
2760
+ }>;
2761
+ answer: string;
2762
+ prompt: string;
2763
+ }
2764
+ export interface LoopJudgeWork {
2765
+ run: LoopJudgeRun;
2766
+ items: LoopJudgeWorkItem[];
2767
+ remaining: number;
2768
+ }
2769
+ export type LoopJudgeRunItemStatus = 'pending' | 'scored' | 'failed';
2770
+ /** What happened to one conversation in one run. */
2771
+ export interface LoopJudgeRunItem {
2772
+ trace_id: string;
2773
+ status: LoopJudgeRunItemStatus;
2774
+ /**
2775
+ * Why this one could not be scored, as the caller reported it. This is where
2776
+ * a wrong model name or a refused key shows up; null for anything that is
2777
+ * not failed.
2778
+ */
2779
+ error: string | null;
2780
+ scored_at: string | null;
2781
+ }
2782
+ export interface LoopJudgeRunItems {
2783
+ run: LoopJudgeRun;
2784
+ items: LoopJudgeRunItem[];
2785
+ /** True when the page ended before the run did. */
2786
+ has_more: boolean;
2787
+ /** Where to carry on from. Present only when `has_more`. */
2788
+ next_offset?: number;
2789
+ }
2790
+ /** One scored conversation going back. */
2791
+ export interface LoopJudgeVerdict {
2792
+ trace_id: string;
2793
+ /** One score per dimension the rubric asked for, each between 0 and 1. */
2794
+ scores?: Record<string, number>;
2795
+ /** Your own combination of them. Left out, the plain mean is used. */
2796
+ overall?: number;
2797
+ reason?: string;
2798
+ /** Only stored if the judge was created with `write_gold`. */
2799
+ gold?: string;
2800
+ /** Mark an item you could not score, instead of dropping it silently. */
2801
+ error?: string;
2802
+ }
2803
+ export interface LoopJudgeVerdictResult {
2804
+ run: LoopJudgeRun;
2805
+ recorded: number;
2806
+ failed: number;
2807
+ rejected: Array<{
2808
+ trace_id: string;
2809
+ error: string;
2810
+ }>;
2811
+ }
2812
+ export interface LoopStats {
2813
+ traces: number;
2814
+ signalled_traces: number;
2815
+ signals: number;
2816
+ by_verdict: Record<string, number>;
2817
+ by_source: Record<string, number>;
2818
+ datasets: number;
2819
+ /** Upper bounds: deduplication runs when a set is built. */
2820
+ ready: Record<LoopMethod, number>;
2821
+ }
2822
+ export interface LoopCandidateParams {
2823
+ completion?: string;
2824
+ /** An alternative that is itself a tool call. */
2825
+ tool_calls?: LoopToolCall[];
2826
+ /** Which model produced this alternative. The teacher, when distilling. */
2827
+ model?: string;
2828
+ model_version?: string;
2829
+ /** Unscored alternatives cannot pair -- a missing score is not a low one. */
2830
+ score?: number;
2831
+ /** Required alongside a score: human | verifier | judge | behavioural. */
2832
+ score_source?: LoopSource;
2833
+ reason?: string;
2834
+ metadata?: Record<string, unknown>;
2835
+ }
2836
+ export interface LoopCandidate {
2837
+ id: string;
2838
+ trace_id: string;
2839
+ model: string | null;
2840
+ model_version: string | null;
2841
+ completion: string;
2842
+ tool_calls?: LoopToolCall[];
2843
+ score: number | null;
2844
+ score_source: LoopSource | null;
2845
+ reason: string | null;
2846
+ metadata: Record<string, unknown>;
2847
+ created_at: string;
2848
+ }
2849
+ /**
2850
+ * The deterministic checks a grader can perform.
2851
+ *
2852
+ * `matches_gold` is the one kind with no expected value of its own: it
2853
+ * compares each answer to the gold answer recorded on that same conversation
2854
+ * (or the human correction when there is no gold), scores sampled
2855
+ * alternatives against the same gold, and is NOT APPLIED to a conversation
2856
+ * that carries neither, so it never marks down an unlabelled answer.
2857
+ */
2858
+ export type LoopGraderKind = 'matches_gold' | 'exact_match' | 'contains' | 'regex' | 'json_valid' | 'numeric' | 'tool_called';
2859
+ export interface LoopGraderConfig {
2860
+ expected?: string;
2861
+ /** Every phrase that must appear. */
2862
+ required?: string[];
2863
+ /** Phrases that must not. On their own these are passed by silence. */
2864
+ forbidden?: string[];
2865
+ pattern?: string;
2866
+ /** Keys the answer must carry, for `json_valid`. */
2867
+ keys?: string[];
2868
+ /**
2869
+ * How far from `expected` still counts, for `numeric`. For `matches_gold`,
2870
+ * how far from the gold answer still counts when both are bare numbers.
2871
+ */
2872
+ tolerance?: number;
2873
+ /** The function that must have been called, for `tool_called`. */
2874
+ function?: string;
2875
+ case_sensitive?: boolean;
2876
+ }
2877
+ export interface LoopGraderParams {
2878
+ name: string;
2879
+ kind: LoopGraderKind;
2880
+ config: LoopGraderConfig;
2881
+ /** The multiplier. Twice as important, twice the weight. Must exceed zero. */
2882
+ weight?: number;
2883
+ enabled?: boolean;
2884
+ /** Limit the rule to one capture source. Omit to apply it everywhere. */
2885
+ deployment_id?: string;
2886
+ /**
2887
+ * Aim the rule at one slice of the corpus. A master group covers its
2888
+ * children. Outside that slice the rule is NOT APPLIED, rather than failed,
2889
+ * so it stays out of the score entirely. Omit to apply it everywhere.
2890
+ */
2891
+ label?: string;
2892
+ }
2893
+ export interface LoopGrader extends LoopGraderParams {
2894
+ id: string;
2895
+ workspace_id: string;
2896
+ weight: number;
2897
+ enabled: boolean;
2898
+ created_by: string | null;
2899
+ created_at: string;
2900
+ updated_at: string;
2901
+ }
2902
+ export interface LoopGraderResult {
2903
+ grader_id: string;
2904
+ name: string;
2905
+ kind: string;
2906
+ weight: number;
2907
+ passed: boolean;
2908
+ score: number;
2909
+ /** What happened, in words -- "failed" alone sends you to read the rule. */
2910
+ detail: string;
2911
+ /** The rule had no opinion. Excluded from the score rather than counted 0. */
2912
+ skipped: boolean;
2913
+ }
2914
+ export interface LoopGradeReport {
2915
+ /** Weighted mean over the rules that applied, 0 to 1. */
2916
+ score: number;
2917
+ results: LoopGraderResult[];
2918
+ /** The denominator. 1.0 from one rule is not 1.0 from six. */
2919
+ applied: number;
2920
+ skipped: number;
2921
+ }
2922
+ export interface LoopGradeResult {
2923
+ report: LoopGradeReport;
2924
+ /** Null when no rule applied, because no verdict was invented. */
2925
+ signal_id: string | null;
2926
+ candidates_graded: number;
2927
+ }
2928
+ export interface LoopLabel {
2929
+ label: string;
2930
+ /** The master group this label belongs to, if any. */
2931
+ parent: string | null;
2932
+ created_by: string | null;
2933
+ created_at: string;
2934
+ /** The dimension this value belongs to. `tag` for a bare name. */
2935
+ key: string;
2936
+ }
2937
+ /** One master group, and how many of a label's conversations are in it. */
2938
+ export interface LoopLabelParentCount {
2939
+ parent: string;
2940
+ traces: number;
2941
+ }
2942
+ export interface LoopLabelCount {
2943
+ label: string;
2944
+ /**
2945
+ * The master group ALL of this label's conversations are in, and nothing
2946
+ * else. Null when they disagree: the registry used to answer the
2947
+ * alphabetically last parent over a mixed group, so `billing` was reported
2948
+ * under `support` while seven of its twelve conversations were in no group
2949
+ * at all. Render "(in x)" from this field only.
2950
+ */
2951
+ parent: string | null;
2952
+ /**
2953
+ * Every master group ANY of them are in, sorted, with how many of this
2954
+ * label's conversations are in each. Empty when there are none.
2955
+ *
2956
+ * Read this, not `parent`, to discover which master groups exist: a label
2957
+ * whose conversations disagree still belongs partly to a real group, and
2958
+ * `parent` is null for it, so a picker built from `parent` alone loses the
2959
+ * group along with the false claim. The count is per group rather than the
2960
+ * label's own total, because adding a label's whole count to its parent is
2961
+ * the same mistake one level down. What is in no group at all is `traces`
2962
+ * minus the sum of these.
2963
+ */
2964
+ parents: LoopLabelParentCount[];
2965
+ traces: number;
2966
+ key: string;
2967
+ }
2968
+ export type TrainingCadence = 'none' | 'daily' | 'weekly';
2969
+ export type TrainingCombinator = 'and' | 'or';
2970
+ /** What answers the customer's traffic today, and therefore what promotion re-points. */
2971
+ export type ServingKind = 'serverless_slug' | 'deployment';
2972
+ export type TrainingRulePausedReason = 'consent_invalid' | 'insufficient_funds' | 'authorizer_not_member' | 'model_not_trainable' | 'platform_capacity'
2973
+ /** A peer was unreachable when the rule was saved, so it has not been validated yet. */
2974
+ | 'validation_pending';
2975
+ /**
2976
+ * How a training rule trains, on the wire of the automatic-training routes.
2977
+ *
2978
+ * DELIBERATELY NOT {@link TrainingMethod}. That union -- 'sft' | 'pt' -- is
2979
+ * training-service's, and the two services enumerate different things: a
2980
+ * training job can be a plain pre-training run, and a training rule cannot,
2981
+ * while a rule may ask for preference training and a job asks for that through
2982
+ * a different field. The values here are the loop service's own constants
2983
+ * (TrainingMethodSFT / TrainingMethodRLHF in
2984
+ * services/loop-service/cmd/training_types.go), which is what these routes
2985
+ * accept and return. Sharing the training-service union here advertised 'pt',
2986
+ * which this service refuses, and made 'rlhf', which it returns, unspellable.
2987
+ *
2988
+ * `rlhf_type` stays a plain string on purpose: the platform refuses it until
2989
+ * capabilities enable preference training, and an SDK union would have to be
2990
+ * republished to keep up with a server-side capability flag.
2991
+ */
2992
+ export type LoopTrainingMethod = 'sft' | 'rlhf';
2993
+ export type TrainType = 'lora' | 'qlora' | 'full';
2994
+ export type GradersScope = 'source' | 'all' | 'none';
2995
+ export type TrainingTrigger = 'rows' | 'cadence' | 'both' | 'manual';
2996
+ export type TrainingRunState = 'built' | 'registering' | 'dataset_validating' | 'preflighting' | 'job_creating' | 'training' | 'checkpoint_ready' | 'candidate_booking' | 'candidate_running' | 'evaluating' | 'reported' | 'awaiting_review' | 'promoting' | 'waiting_funds' | 'promoted' | 'rejected' | 'expired' | 'failed' | 'budget_stopped' | 'cancelled' | 'superseded' | 'rolled_back';
2997
+ /** The closed set a run never leaves. */
2998
+ export declare const TERMINAL_RUN_STATES: readonly TrainingRunState[];
2999
+ export type TrainingVerdict = 'better' | 'not_better' | 'inconclusive' | 'not_evaluated';
3000
+ export type TrainingDecision = 'auto_promoted' | 'promoted' | 'rejected' | 'auto_rejected' | 'rolled_back';
3001
+ export type NotifyState = 'none' | 'pending' | 'sending' | 'sent' | 'suppressed' | 'failed';
3002
+ export type EvaluationStatus = 'open' | 'done' | 'failed' | 'budget_stopped';
3003
+ export type EvaluationItemStatus = 'pending' | 'working' | 'done' | 'failed';
3004
+ export type EvaluationWinner = 'candidate' | 'incumbent' | 'tie';
3005
+ /** Typed warning codes. Each surface renders its own plain sentence. */
3006
+ export type EvaluationWarningCode = 'holdout_too_small' | 'judge_unreliable' | 'graders_not_applied' | 'many_failures' | 'no_judge' | 'judge_labelled_training_rows' | 'capture_source_moved';
3007
+ export type ConsentVia = 'console' | 'api_key' | 'sdk' | 'mcp';
3008
+ /** A function an assistant turn asked to invoke. `arguments` is JSON-encoded. */
3009
+ export interface TrainingToolCall {
3010
+ id?: string;
3011
+ type?: string;
3012
+ function: {
3013
+ name: string;
3014
+ arguments: string;
3015
+ };
3016
+ }
3017
+ /** One conversational turn, in the chat-completions shape. */
3018
+ export interface TrainingMessage {
3019
+ role: string;
3020
+ content?: string;
3021
+ tool_calls?: TrainingToolCall[];
3022
+ tool_call_id?: string;
3023
+ name?: string;
3024
+ }
3025
+ /** What serves this model today. The name is also the alias a promotion writes. */
3026
+ export interface TrainingRuleServing {
3027
+ kind: ServingKind;
3028
+ name: string;
3029
+ /** Set only when kind is 'deployment'. */
3030
+ deployment_id: string | null;
3031
+ }
3032
+ /** One rung of a ranked GPU ladder. The provider is the neutral public brand. */
3033
+ export interface TrainingGPURung {
3034
+ gpu_type: string;
3035
+ gpu_count: number;
3036
+ provider: string;
3037
+ region: string;
3038
+ tier: string;
3039
+ }
3040
+ /** The consent object: one standing instruction to train, judge and possibly promote. */
3041
+ export interface TrainingRule {
3042
+ id: string;
3043
+ workspace_id: string;
3044
+ build_rule_id: string;
3045
+ name: string;
3046
+ enabled: boolean;
3047
+ /** Why the platform stopped firing this rule. Waiting must not look like broken. */
3048
+ paused_reason: TrainingRulePausedReason | null;
3049
+ cadence: TrainingCadence;
3050
+ cadence_hour_utc: number;
3051
+ cadence_weekday: number | null;
3052
+ combinator: TrainingCombinator;
3053
+ min_new_rows: number | null;
3054
+ /** Spreads firings across the hour so every daily rule does not land on one minute. */
3055
+ jitter_seconds: number;
3056
+ next_due_at: string | null;
3057
+ last_checked_at: string | null;
3058
+ last_fired_at: string | null;
3059
+ last_run_id: string | null;
3060
+ /** Rendered verbatim: "waiting: ...", "due, but ...", "fired: run 7 from ...". */
3061
+ last_reason: string | null;
3062
+ serving: TrainingRuleServing;
3063
+ model_id: string;
3064
+ /** A 40-hex commit, pinned when the rule is saved. */
3065
+ model_revision: string;
3066
+ training_method: LoopTrainingMethod;
3067
+ /** Refused by the platform until capabilities enable preference training. */
3068
+ rlhf_type: string | null;
3069
+ train_type: TrainType;
3070
+ config: Record<string, unknown>;
3071
+ train_gpu_priorities: TrainingGPURung[];
3072
+ train_max_price_hour_cents: number;
3073
+ deploy_gpu_priorities: TrainingGPURung[];
3074
+ deploy_max_price_hour_cents: number;
3075
+ training_ceiling_cents: number;
3076
+ candidate_ceiling_cents: number;
3077
+ /** A dollar figure, never a call count. */
3078
+ eval_ceiling_cents: number;
3079
+ monthly_ceiling_cents: number | null;
3080
+ eval_judge_id: string | null;
3081
+ eval_judge_model: string | null;
3082
+ eval_graders_scope: GradersScope;
3083
+ eval_max_rows: number;
3084
+ eval_max_tokens: number;
3085
+ min_holdout_rows: number;
3086
+ /**
3087
+ * The standing benchmark replayed on every run of this rule, beside the
3088
+ * per-run comparison and never instead of it. Null is the ordinary state.
3089
+ *
3090
+ * Set with `setTrainingRuleBenchmark`, not by an update to the rule: it is a
3091
+ * decision to replay a fixed set on every future run, and it has refusals of
3092
+ * its own. It raises no amount, so it does not invalidate consent.
3093
+ */
3094
+ benchmark_id: string | null;
3095
+ auto_promote: boolean;
3096
+ promote_margin: number;
3097
+ promote_min_win_rate: number;
3098
+ /** A gate, not a badge. */
3099
+ min_judge_agreement: number;
3100
+ review_window_hours: number;
3101
+ candidate_boot_deadline_minutes: number;
3102
+ keep_candidate_warm_minutes: number;
3103
+ /** The member whose wallet pays, and the identity every peer call is stamped with. */
3104
+ authorized_by: string;
3105
+ terms_version: string;
3106
+ revision: number;
3107
+ /** Equal to revision only while the consent is current. */
3108
+ accepted_revision: number;
3109
+ accepted_at: string | null;
3110
+ deleted_at: string | null;
3111
+ created_at: string;
3112
+ updated_at: string;
3113
+ }
3114
+ /** One append-only record of a member agreeing to spend, with the words they read. */
3115
+ export interface TrainingRuleConsent {
3116
+ id: number;
3117
+ rule_id: string;
3118
+ workspace_id: string;
3119
+ revision: number;
3120
+ terms_version: string;
3121
+ terms_text: string;
3122
+ snapshot: Record<string, unknown>;
3123
+ accepted_by: string;
3124
+ accepted_via: ConsentVia;
3125
+ accepted_at: string;
3126
+ }
3127
+ export interface TrainingRulePreflightRefusal {
3128
+ stage: string;
3129
+ code: string;
3130
+ message: string;
3131
+ }
3132
+ /** One key that cannot follow a cutover, named so the caller can say which. */
3133
+ export interface TrainingRuleKeyRef {
3134
+ key_id: string;
3135
+ prefix: string;
3136
+ name: string;
3137
+ }
3138
+ export interface TrainingRuleKeyCheck {
3139
+ keys_missing_deployments_read: TrainingRuleKeyRef[];
3140
+ }
3141
+ /** The side-effect-free estimate, and the exact sentence the member will accept. */
3142
+ export interface TrainingRulePreflight {
3143
+ valid: boolean;
3144
+ model_revision: string;
3145
+ worst_hourly_training_cents: number;
3146
+ worst_hourly_candidate_cents: number;
3147
+ max_training_hours: number;
3148
+ max_candidate_hours: number;
3149
+ /** Informational. The fence on evaluation is the dollar ceiling, never this. */
3150
+ eval_calls_max: number;
3151
+ training_ceiling_cents: number;
3152
+ candidate_ceiling_cents: number;
3153
+ eval_ceiling_cents: number;
3154
+ warnings: string[];
3155
+ refusals: TrainingRulePreflightRefusal[];
3156
+ /**
3157
+ * Every peer the platform could not reach on this pass, one entry each, in
3158
+ * the same `{stage, code, message}` shape as a refusal.
3159
+ *
3160
+ * A warning, not a refusal: an unreachable peer judged nothing, so it never
3161
+ * refuses a save. It is still why the rule cannot fire — both
3162
+ * `worst_hourly_*` come back `0` — so `valid` is `false`. Key a "cannot
3163
+ * save" state on this when `refusals` is empty; the same sentences are also
3164
+ * in `warnings`, so render one or the other.
3165
+ *
3166
+ * Key it on THIS, not on `model_revision`. A degraded pass echoes back the
3167
+ * `model_revision` the request carried rather than emptying it, so
3168
+ * `if (!model_revision)` is false on exactly the passes it was meant to
3169
+ * catch. Use `unreachable.length > 0`, or `/^[0-9a-f]{40}$/`.
3170
+ */
3171
+ unreachable: TrainingRulePreflightRefusal[];
3172
+ key_check: TrainingRuleKeyCheck;
3173
+ terms_text: string;
3174
+ terms_version: string;
3175
+ }
3176
+ export interface TrainingRuleBuildSpec {
3177
+ method: string;
3178
+ spec: Record<string, unknown>;
3179
+ }
3180
+ /**
3181
+ * The editable half of a build rule.
3182
+ *
3183
+ * Deliberately not `LoopBuildRuleParams`: that type requires `name` and
3184
+ * `method`, which the PUT route does not accept, so an edit written against it
3185
+ * would have to resend two fields the platform ignores and a caller could not
3186
+ * tell that changing them did nothing.
3187
+ */
3188
+ export interface LoopBuildRuleUpdateParams {
3189
+ enabled?: boolean;
3190
+ /** Rows reviewed SINCE THE LAST BUILD before this fires again. Minimum 10. */
3191
+ min_new_rows?: number;
3192
+ /** The same selection `createDataset` takes, replayed verbatim. */
3193
+ spec?: LoopDatasetCreateParams;
3194
+ }
3195
+ /**
3196
+ * `?` means absent: leave that part of the sentence alone. `null` is only
3197
+ * allowed where the column is nullable and clearing it is a real edit, and it
3198
+ * is spelled out field by field rather than applied to the whole object.
3199
+ */
3200
+ export interface TrainingRuleTriggerInput {
3201
+ cadence?: TrainingCadence;
3202
+ cadence_hour_utc?: number;
3203
+ /** null drops the weekday, which is what a weekly rule moved to daily needs. */
3204
+ cadence_weekday?: number | null;
3205
+ combinator?: TrainingCombinator;
3206
+ /** null removes the row floor, leaving the schedule as the only trigger. */
3207
+ min_new_rows?: number | null;
3208
+ }
3209
+ export interface TrainingRuleTrainingInput {
3210
+ model_id?: string;
3211
+ model_revision?: string;
3212
+ training_method?: LoopTrainingMethod;
3213
+ /**
3214
+ * A plain string, refused by the platform until capabilities enable
3215
+ * preference training. null clears it, which is what moving a rule back to
3216
+ * plain supervised training means.
3217
+ */
3218
+ rlhf_type?: string | null;
3219
+ train_type?: TrainType;
3220
+ config?: Record<string, unknown>;
3221
+ train_gpu_priorities?: TrainingGPURung[];
3222
+ train_max_price_hour_cents?: number;
3223
+ }
3224
+ export interface TrainingRuleDeployInput {
3225
+ deploy_gpu_priorities?: TrainingGPURung[];
3226
+ deploy_max_price_hour_cents?: number;
3227
+ context_length?: number;
3228
+ quant?: string;
3229
+ serving_config?: Record<string, unknown>;
3230
+ hf_integration_id?: string;
3231
+ }
3232
+ export interface TrainingRuleMoneyInput {
3233
+ training_ceiling_cents?: number;
3234
+ candidate_ceiling_cents?: number;
3235
+ eval_ceiling_cents?: number;
3236
+ /** null removes the monthly cap. Absent leaves it exactly where it stands. */
3237
+ monthly_ceiling_cents?: number | null;
3238
+ eval_max_rows?: number;
3239
+ }
3240
+ export interface TrainingRuleEvaluationInput {
3241
+ /** null removes the judge from this rule. */
3242
+ judge_id?: string | null;
3243
+ /** null drops the override, returning to the judge's own advisory model. */
3244
+ judge_model?: string | null;
3245
+ graders_scope?: GradersScope;
3246
+ min_holdout_rows?: number;
3247
+ eval_max_tokens?: number;
3248
+ }
3249
+ export interface TrainingRulePromotionInput {
3250
+ auto_promote?: boolean;
3251
+ promote_margin?: number;
3252
+ promote_min_win_rate?: number;
3253
+ min_judge_agreement?: number;
3254
+ review_window_hours?: number;
3255
+ keep_candidate_warm_minutes?: number;
3256
+ }
3257
+ /**
3258
+ * The version of the terms the member read and accepted.
3259
+ *
3260
+ * Send back the `terms_version` the preflight returned. The SDK deliberately
3261
+ * ships no constant for it: a pinned version in a published package goes stale
3262
+ * the moment the platform revises the terms, and the value a member consented
3263
+ * to has to be the one they were actually shown.
3264
+ */
3265
+ export interface TrainingRuleAcceptTerms {
3266
+ terms_version: string;
3267
+ }
3268
+ /** The create body without the yes. The preflight route takes exactly this. */
3269
+ export interface TrainingRulePreflightRequest {
3270
+ workspace_id?: string;
3271
+ name?: string;
3272
+ /** Exactly one of build_rule_id and build is given. */
3273
+ build_rule_id?: string;
3274
+ build?: TrainingRuleBuildSpec;
3275
+ trigger?: TrainingRuleTriggerInput;
3276
+ serving?: TrainingRuleServing;
3277
+ training?: TrainingRuleTrainingInput;
3278
+ deploy?: TrainingRuleDeployInput;
3279
+ money?: TrainingRuleMoneyInput;
3280
+ evaluation?: TrainingRuleEvaluationInput;
3281
+ promotion?: TrainingRulePromotionInput;
3282
+ enabled?: boolean;
3283
+ }
3284
+ export interface TrainingRuleCreateRequest extends TrainingRulePreflightRequest {
3285
+ /** Absent is a refusal, not a default. */
3286
+ accept_terms?: TrainingRuleAcceptTerms;
3287
+ /**
3288
+ * The standing benchmark this rule replays, set here so it applies to the
3289
+ * rule's FIRST run.
3290
+ *
3291
+ * A rule can fire within seconds of being created, and every step of a run
3292
+ * reads the benchmark off the rule as it stood at that moment. Attaching one
3293
+ * afterwards with `setTrainingRuleBenchmark` applies from the next run and
3294
+ * says nothing about the first, which is the run somebody is watching.
3295
+ *
3296
+ * Refused on the same terms the attach route refuses it: `404` when it is not
3297
+ * this workspace's, `409 BENCHMARK_RETIRED` when it has been retired. Use
3298
+ * `setTrainingRuleBenchmark` to change or remove it later; it is not on the
3299
+ * preflight body and not on the update body.
3300
+ */
3301
+ benchmark_id?: string;
3302
+ }
3303
+ export interface TrainingRuleUpdateRequest extends TrainingRulePreflightRequest {
3304
+ expected_revision?: number;
3305
+ accept_terms?: TrainingRuleAcceptTerms;
3306
+ }
3307
+ export interface TrainingRuleConsentRequest {
3308
+ terms_version: string;
3309
+ revision: number;
3310
+ }
3311
+ export interface TrainingRuleListParams {
3312
+ enabled?: boolean;
3313
+ limit?: number;
3314
+ offset?: number;
3315
+ }
3316
+ export interface TrainingRunPromoteRequest {
3317
+ expected_revision?: number;
3318
+ /** Required to promote an inconclusive comparison. */
3319
+ force?: boolean;
3320
+ }
3321
+ export interface TrainingRunRejectRequest {
3322
+ reason?: string;
3323
+ }
3324
+ export interface TrainingRunRollbackRequest {
3325
+ reason?: string;
3326
+ }
3327
+ export interface TrainingRunCancelRequest {
3328
+ reason?: string;
3329
+ }
3330
+ export interface TrainingRunListParams {
3331
+ rule_id?: string;
3332
+ state?: TrainingRunState;
3333
+ limit?: number;
3334
+ offset?: number;
3335
+ }
3336
+ /** One firing, from the curated set to the decision. Money is frozen at fire time. */
3337
+ export interface TrainingRun {
3338
+ id: string;
3339
+ rule_id: string | null;
3340
+ rule_name: string | null;
3341
+ workspace_id: string;
3342
+ seq: number;
3343
+ rule_revision: number;
3344
+ authorized_by: string;
3345
+ rule_snapshot: Record<string, unknown>;
3346
+ trigger: TrainingTrigger;
3347
+ fired_reason: string | null;
3348
+ state: TrainingRunState;
3349
+ state_entered_at: string;
3350
+ attempts: number;
3351
+ next_poll_at: string | null;
3352
+ last_reason: string | null;
3353
+ error_code: string | null;
3354
+ last_error: string | null;
3355
+ training_ceiling_cents: number;
3356
+ candidate_ceiling_cents: number;
3357
+ eval_ceiling_cents: number;
3358
+ billed_training_cents: number;
3359
+ billed_candidate_cents: number;
3360
+ spent_eval_cents: number;
3361
+ loop_dataset_id: string | null;
3362
+ train_rows: number | null;
3363
+ holdout_rows: number | null;
3364
+ train_dataset_id: string | null;
3365
+ holdout_dataset_id: string | null;
3366
+ training_job_id: string | null;
3367
+ job_status: string | null;
3368
+ queue_deadline_at: string | null;
3369
+ checkpoint_id: string | null;
3370
+ checkpoint_step: number | null;
3371
+ training_eval_loss: number | null;
3372
+ candidate_deployment_id: string | null;
3373
+ candidate_name: string | null;
3374
+ candidate_seq: number;
3375
+ candidate_status: string | null;
3376
+ candidate_deadline_at: string | null;
3377
+ candidate_cleanup_at: string | null;
3378
+ evaluation_id: string | null;
3379
+ /** This run's replay of the rule's standing benchmark, if it had one. */
3380
+ benchmark_run_id: string | null;
3381
+ /**
3382
+ * What the replay's calls cost. Kept apart from `spent_eval_cents` so a
3383
+ * report can say what the comparison cost and what the benchmark cost; both
3384
+ * come out of `eval_ceiling_cents` and their sum can never exceed it, so
3385
+ * anything totalling what a run cost has to add this one too.
3386
+ */
3387
+ benchmark_spent_cents: number;
3388
+ alias_name: string | null;
3389
+ alias_written_at: string | null;
3390
+ serving_before: TrainingRuleServing | null;
3391
+ rollback_available_until: string | null;
3392
+ verdict: TrainingVerdict | null;
3393
+ decision: TrainingDecision | null;
3394
+ decided_by: string | null;
3395
+ decided_at: string | null;
3396
+ auto_promote_at: string | null;
3397
+ review_deadline_at: string | null;
3398
+ notify_state: NotifyState;
3399
+ notify_event: string | null;
3400
+ notify_error: string | null;
3401
+ notify_attempts: number;
3402
+ created_at: string;
3403
+ updated_at: string;
3404
+ finished_at: string | null;
3405
+ }
3406
+ /** One row of the Runs table, carrying everything the list draws. */
3407
+ export interface TrainingRunSummary {
3408
+ id: string;
3409
+ rule_id: string | null;
3410
+ rule_name: string | null;
3411
+ workspace_id: string;
3412
+ seq: number;
3413
+ trigger: TrainingTrigger;
3414
+ state: TrainingRunState;
3415
+ state_entered_at: string;
3416
+ last_reason: string | null;
3417
+ error_code: string | null;
3418
+ verdict: TrainingVerdict | null;
3419
+ decision: TrainingDecision | null;
3420
+ train_rows: number | null;
3421
+ holdout_rows: number | null;
3422
+ billed_training_cents: number;
3423
+ billed_candidate_cents: number;
3424
+ spent_eval_cents: number;
3425
+ /** The fourth money column. A row that leaves it out adds up short. */
3426
+ benchmark_spent_cents: number;
3427
+ training_job_id: string | null;
3428
+ candidate_deployment_id: string | null;
3429
+ evaluation_id: string | null;
3430
+ review_deadline_at: string | null;
3431
+ auto_promote_at: string | null;
3432
+ notify_state: NotifyState;
3433
+ created_at: string;
3434
+ finished_at: string | null;
3435
+ }
3436
+ /** One line of the timeline. detail holds codes, ids and cents, never prompt text. */
3437
+ export interface TrainingRunEvent {
3438
+ seq: number;
3439
+ ts: string;
3440
+ from_state: TrainingRunState | null;
3441
+ to_state: TrainingRunState;
3442
+ reason: string;
3443
+ error_code: string | null;
3444
+ detail: Record<string, unknown>;
3445
+ actor: string;
3446
+ }
3447
+ export interface TrainingRunLinks {
3448
+ training_job_url: string | null;
3449
+ candidate_url: string | null;
3450
+ dataset_url: string | null;
3451
+ evaluation_url: string | null;
3452
+ /** The TREND the benchmark number belongs to, not the one replay. */
3453
+ benchmark_url: string | null;
3454
+ }
3455
+ /** What this reader may do right now. A button that cannot work is never shown. */
3456
+ export interface TrainingRunActions {
3457
+ promote: boolean;
3458
+ reject: boolean;
3459
+ rollback: boolean;
3460
+ cancel: boolean;
3461
+ }
3462
+ export interface EvaluationRef {
3463
+ kind: ServingKind;
3464
+ name: string;
3465
+ deployment_id: string | null;
3466
+ model_version: string | null;
3467
+ }
3468
+ /** Frozen on the evaluation. Both sides are regenerated with identical decoding. */
3469
+ export interface EvaluationDecoding {
3470
+ temperature: number;
3471
+ max_tokens: number;
3472
+ stream: boolean;
3473
+ }
3474
+ export interface EvaluationJudgeDimension {
3475
+ key: string;
3476
+ description: string;
3477
+ }
3478
+ export interface EvaluationDimensionScore {
3479
+ dimension: string;
3480
+ incumbent: number;
3481
+ candidate: number;
3482
+ delta: number;
3483
+ }
3484
+ export interface EvaluationGraderScore {
3485
+ grader_id: string;
3486
+ name: string;
3487
+ incumbent: number;
3488
+ candidate: number;
3489
+ delta: number;
3490
+ /** A grader that applied to four rows has not measured anything. */
3491
+ items_applied: number;
3492
+ }
3493
+ export interface EvaluationWarning {
3494
+ code: EvaluationWarningCode;
3495
+ message: string;
3496
+ }
3497
+ /** The thresholds this verdict was measured against, frozen with the report. */
3498
+ export interface EvaluationMargin {
3499
+ promote_margin: number;
3500
+ promote_min_win_rate: number;
3501
+ min_judge_agreement: number;
3502
+ min_holdout_rows: number;
3503
+ }
3504
+ /**
3505
+ * One rubric dimension of a judge-versus-human agreement.
3506
+ *
3507
+ * Deliberately not EvaluationDimensionScore: that type carries `incumbent` and
3508
+ * `candidate`, which are the two models being compared, and an agreement has
3509
+ * neither. `pairs` is per dimension because a reviewer who rated one dimension
3510
+ * and skipped another leaves a different denominator behind each number.
3511
+ */
3512
+ export interface JudgeAgreementDimension {
3513
+ dimension: string;
3514
+ pairs: number;
3515
+ agreement: number | null;
3516
+ mean_abs_error: number | null;
3517
+ }
3518
+ /** How often this judge agreed with the workspace's own reviewers. */
3519
+ export interface JudgeAgreement {
3520
+ judge_id: string;
3521
+ pairs: number;
3522
+ agreement: number | null;
3523
+ mean_abs_error: number | null;
3524
+ per_dimension: JudgeAgreementDimension[];
3525
+ window_days: number;
3526
+ computed_at: string;
3527
+ /** "Not enough reviewer overlap yet" is an answer; 100% of two is not. */
3528
+ enough_pairs: boolean;
3529
+ }
3530
+ /** The comparison report: the candidate against what serves today, same rows. */
3531
+ export interface Evaluation {
3532
+ id: string;
3533
+ run_id: string;
3534
+ workspace_id: string;
3535
+ loop_dataset_id: string | null;
3536
+ split: string;
3537
+ incumbent_ref: EvaluationRef;
3538
+ candidate_ref: EvaluationRef;
3539
+ judge_id: string | null;
3540
+ judge_name: string | null;
3541
+ judge_model: string | null;
3542
+ judge_instructions: string | null;
3543
+ judge_dimensions: EvaluationJudgeDimension[];
3544
+ judge_system_prompt: string | null;
3545
+ grader_ids: string[];
3546
+ decoding: EvaluationDecoding;
3547
+ rows_selected: number;
3548
+ rows_scored: number;
3549
+ rows_failed: number;
3550
+ /** win_rate is wins / (wins + losses). Ties are excluded and reported separately. */
3551
+ wins: number | null;
3552
+ losses: number | null;
3553
+ ties: number | null;
3554
+ win_rate: number | null;
3555
+ incumbent_mean: number | null;
3556
+ candidate_mean: number | null;
3557
+ mean_delta: number | null;
3558
+ judge_incumbent_mean: number | null;
3559
+ judge_candidate_mean: number | null;
3560
+ grader_incumbent_mean: number | null;
3561
+ grader_candidate_mean: number | null;
3562
+ grader_items_applied: number;
3563
+ per_dimension: EvaluationDimensionScore[];
3564
+ per_grader: EvaluationGraderScore[];
3565
+ judge_agreement: JudgeAgreement | null;
3566
+ /** Informational: there is no incumbent counterpart to compare it against. */
3567
+ trainer_eval_loss: number | null;
3568
+ warnings: EvaluationWarning[];
3569
+ verdict: TrainingVerdict | null;
3570
+ verdict_reason: string | null;
3571
+ margin_used: EvaluationMargin | null;
3572
+ prompt_tokens: number;
3573
+ completion_tokens: number;
3574
+ spent_cents: number;
3575
+ status: EvaluationStatus;
3576
+ last_error: string | null;
3577
+ created_at: string;
3578
+ finished_at: string | null;
3579
+ }
3580
+ /** One model's answer to one held-out conversation, with the scores it earned. */
3581
+ export interface EvaluationItemSide {
3582
+ completion: string | null;
3583
+ tool_calls: TrainingToolCall[] | null;
3584
+ grader: Record<string, unknown> | null;
3585
+ judge: Record<string, unknown> | null;
3586
+ score: number | null;
3587
+ model_version: string | null;
3588
+ }
3589
+ /**
3590
+ * One paired conversation behind the numbers.
3591
+ *
3592
+ * trace_id is nullable on purpose: deleting one conversation must not shrink a
3593
+ * finished report so that its stated n and its visible rows disagree.
3594
+ */
3595
+ export interface EvaluationItem {
3596
+ id: number;
3597
+ dataset_item_id: number;
3598
+ trace_id: string | null;
3599
+ trace_url: string | null;
3600
+ prompt: TrainingMessage[];
3601
+ tools: unknown[] | null;
3602
+ incumbent: EvaluationItemSide;
3603
+ candidate: EvaluationItemSide;
3604
+ human_verdict: number | null;
3605
+ winner: EvaluationWinner | null;
3606
+ status: EvaluationItemStatus;
3607
+ error: string | null;
3608
+ }
3609
+ export interface EvaluationItemListParams {
3610
+ winner?: EvaluationWinner;
3611
+ /** Capped at 100 by the service. */
3612
+ limit?: number;
3613
+ offset?: number;
3614
+ }
3615
+ export interface JudgeAgreementParams {
3616
+ /** RFC 3339. Narrows the window the agreement is computed over. */
3617
+ from?: string;
3618
+ to?: string;
3619
+ }
3620
+ /** Which model, whose words, how much. The cap is pushed to the workspace key. */
3621
+ export interface AgentSettings {
3622
+ workspace_id: string;
3623
+ default_model: string | null;
3624
+ judge_system_prompt: string | null;
3625
+ sampler_system_prompt: string | null;
3626
+ eval_monthly_cap_cents: number | null;
3627
+ updated_by: string | null;
3628
+ updated_at: string;
3629
+ }
3630
+ /** Absent leaves a setting alone; a present null returns it to the platform default. */
3631
+ export interface AgentSettingsRequest {
3632
+ default_model?: string | null;
3633
+ judge_system_prompt?: string | null;
3634
+ sampler_system_prompt?: string | null;
3635
+ eval_monthly_cap_cents?: number | null;
3636
+ }
3637
+ /** A re-pointable public handle. Promotion is one row write here. */
3638
+ export interface InferenceAlias {
3639
+ workspace_id: string;
3640
+ name: string;
3641
+ target_inference_id: string;
3642
+ previous_target_inference_id: string | null;
3643
+ /** Set only at adoption, when the alias takes over a deployment's own name. */
3644
+ shadows_inference_id: string | null;
3645
+ set_by: string;
3646
+ origin: string | null;
3647
+ created_at: string;
3648
+ updated_at: string;
3649
+ }
3650
+ export interface InferenceAliasRequest {
3651
+ target_inference_id: string;
3652
+ origin?: string;
3653
+ }
3654
+ /**
3655
+ * A benchmark is active until it is retired. There is no delete and no update:
3656
+ * the series of numbers measured against a set is what a benchmark is for, so
3657
+ * an edit would make every number before it incomparable with every number
3658
+ * after it, and a delete throws the series away.
3659
+ */
3660
+ export type BenchmarkStatus = 'active' | 'retired';
3661
+ /** Where the pinned conversations are copied from. Read once, at creation. */
3662
+ export type BenchmarkSourceKind = 'traces' | 'dataset';
3663
+ /**
3664
+ * The life of one replay.
3665
+ *
3666
+ * `budget_stopped` reached the amount left for it inside the run's own ceiling,
3667
+ * and `abandoned` ran out of the time the run sets aside for it. Both are
3668
+ * points on the trend carrying `status_reason`, never silent gaps.
3669
+ */
3670
+ export type BenchmarkRunStatus = 'open' | 'done' | 'failed' | 'budget_stopped' | 'abandoned';
3671
+ /**
3672
+ * How the pinned conversations become one number, frozen on the benchmark.
3673
+ *
3674
+ * `require_all_rows` is what makes the number comparable at all: a mean over
3675
+ * whichever conversations happened to succeed is a measurement of a different
3676
+ * set, so a short replay publishes no score and says why. A partial benchmark
3677
+ * is a missing number, never a lower one.
3678
+ */
3679
+ export interface BenchmarkScoring {
3680
+ metric: string;
3681
+ scale: string;
3682
+ aggregate: string;
3683
+ require_all_rows: boolean;
3684
+ }
3685
+ /**
3686
+ * One rubric dimension of one replay, for both models.
3687
+ *
3688
+ * Deliberately not `EvaluationDimensionScore`: these are absolute means on a
3689
+ * fixed set and those are paired means on that run's own held-back rows.
3690
+ */
3691
+ export interface BenchmarkDimensionScore {
3692
+ dimension: string;
3693
+ incumbent: number;
3694
+ candidate: number;
3695
+ delta: number;
3696
+ }
3697
+ /**
3698
+ * The frozen definition: the oracle, the scoring rule and the fingerprint of
3699
+ * the set. The conversations themselves are their own paged route.
3700
+ */
3701
+ export interface Benchmark {
3702
+ id: string;
3703
+ workspace_id: string;
3704
+ name: string;
3705
+ status: BenchmarkStatus;
3706
+ /** Provenance only: everything needed from the judge is copied below it. */
3707
+ judge_id: string | null;
3708
+ judge_name: string;
3709
+ judge_model: string;
3710
+ judge_instructions: string;
3711
+ judge_dimensions: EvaluationJudgeDimension[];
3712
+ judge_system_prompt: string | null;
3713
+ decoding: EvaluationDecoding;
3714
+ scoring: BenchmarkScoring;
3715
+ item_count: number;
3716
+ /** Recomputed from the rows and compared at the start of every replay. */
3717
+ items_digest: string;
3718
+ /** A cap INSIDE the run's own amount for judge calls, never an addition to it. */
3719
+ per_run_ceiling_cents: number;
3720
+ created_by: string;
3721
+ created_at: string;
3722
+ retired_at: string | null;
3723
+ retired_by: string | null;
3724
+ }
3725
+ /**
3726
+ * One pinned conversation.
3727
+ *
3728
+ * `source_trace_id` is a link and is allowed to go null: the prompt was copied
3729
+ * at pin time, so retention removing the conversation takes away the ability to
3730
+ * open it and takes away nothing else. Every number already measured stays
3731
+ * exactly as comparable as it was.
3732
+ */
3733
+ export interface BenchmarkItem {
3734
+ id: number;
3735
+ ordinal: number;
3736
+ source_trace_id: string | null;
3737
+ source_trace_url: string | null;
3738
+ prompt: TrainingMessage[];
3739
+ tools: unknown[] | null;
3740
+ }
3741
+ /**
3742
+ * One replay: this benchmark, on this training run, against both models.
3743
+ *
3744
+ * IT DOES NOT GATE PROMOTION. The paired comparison applies the judge to both
3745
+ * sides of one conversation in one pass, so most of the judge's variance
3746
+ * cancels in its delta; an absolute mean carries that variance whole. What
3747
+ * ships here is the number, the difference against what serves today on the
3748
+ * same set, and the history. A person reads the trend.
3749
+ */
3750
+ export interface BenchmarkRun {
3751
+ id: string;
3752
+ benchmark_id: string;
3753
+ benchmark_name: string;
3754
+ run_id: string;
3755
+ workspace_id: string;
3756
+ incumbent_ref: EvaluationRef;
3757
+ candidate_ref: EvaluationRef;
3758
+ /** The set actually scored, recorded beside the score. */
3759
+ items_digest: string;
3760
+ rows_total: number;
3761
+ rows_scored: number;
3762
+ rows_failed: number;
3763
+ /** Absolute, on the frozen judge's scale, and null unless every row scored. */
3764
+ candidate_score: number | null;
3765
+ incumbent_score: number | null;
3766
+ score_delta: number | null;
3767
+ per_dimension: BenchmarkDimensionScore[];
3768
+ prompt_tokens: number;
3769
+ completion_tokens: number;
3770
+ spent_cents: number;
3771
+ ceiling_cents: number;
3772
+ status: BenchmarkRunStatus;
3773
+ /** The whole explanation when there is no score, so never empty on one. */
3774
+ status_reason: string | null;
3775
+ created_at: string;
3776
+ finished_at: string | null;
3777
+ }
3778
+ /**
3779
+ * One point on the trend line.
3780
+ *
3781
+ * It carries the checkpoint and what the person then decided, because a trend
3782
+ * with no idea what changed between two points is a chart rather than an
3783
+ * answer.
3784
+ */
3785
+ export interface BenchmarkHistoryPoint {
3786
+ benchmark_run_id: string;
3787
+ run_id: string;
3788
+ run_seq: number;
3789
+ rule_id: string | null;
3790
+ rule_name: string | null;
3791
+ created_at: string;
3792
+ candidate_score: number | null;
3793
+ incumbent_score: number | null;
3794
+ score_delta: number | null;
3795
+ checkpoint_id: string | null;
3796
+ decision: TrainingDecision | null;
3797
+ status: BenchmarkRunStatus;
3798
+ status_reason: string | null;
3799
+ }
3800
+ /** Where the conversations are copied FROM. Read once, at creation, never again. */
3801
+ export interface BenchmarkSource {
3802
+ kind: BenchmarkSourceKind;
3803
+ trace_ids?: string[];
3804
+ dataset_id?: string;
3805
+ split?: string;
3806
+ }
3807
+ /**
3808
+ * The pin: 10 to 200 conversations, and a judge with a rubric to measure them.
3809
+ *
3810
+ * A conversation with nothing to ask a model is refused rather than skipped.
3811
+ * Pinning 47 of the 50 somebody chose is the set being wrong from the first
3812
+ * day, and they would never find out.
3813
+ */
3814
+ export interface BenchmarkCreateParams {
3815
+ name: string;
3816
+ judge_id: string;
3817
+ source: BenchmarkSource;
3818
+ /** Absent uses the judge's own model, and absent that the workspace default. */
3819
+ judge_model?: string;
3820
+ /** What both models are given to answer in. A shorter answer is a different answer. */
3821
+ max_tokens?: number;
3822
+ per_run_ceiling_cents: number;
3823
+ }
3824
+ export interface BenchmarkListParams {
3825
+ status?: BenchmarkStatus;
3826
+ limit?: number;
3827
+ offset?: number;
3828
+ }
3829
+ export interface BenchmarkItemListParams {
3830
+ /** Capped at 100 by the service. */
3831
+ limit?: number;
3832
+ offset?: number;
3833
+ }
3834
+ export interface BenchmarkHistoryParams {
3835
+ limit?: number;
3836
+ offset?: number;
3837
+ }
3838
+ /** A reason is a courtesy here, not a requirement. */
3839
+ export interface BenchmarkRetireParams {
3840
+ reason?: string;
3841
+ }
3842
+ /**
3843
+ * Attach a benchmark to a rule, or detach it with a present null.
3844
+ *
3845
+ * Not optional, and not omittable: this route sets the field, so an absent key
3846
+ * would be a request with nothing in it. Send the id to attach, null to detach.
3847
+ */
3848
+ export interface TrainingRuleBenchmarkRequest {
3849
+ benchmark_id: string | null;
3850
+ }
3851
+ export interface TrainingRuleListResponse {
3852
+ rules: TrainingRule[];
3853
+ total: number;
3854
+ }
3855
+ export interface TrainingRuleResponse {
3856
+ rule: TrainingRule;
3857
+ recent_runs: TrainingRunSummary[];
3858
+ month_spent_cents: number;
3859
+ judge_agreement: JudgeAgreement | null;
3860
+ }
3861
+ export interface TrainingRuleMutationResponse {
3862
+ rule: TrainingRule;
3863
+ /** The rule will not fire again until someone confirms the new amounts. */
3864
+ consent_required: boolean;
3865
+ preflight: TrainingRulePreflight | null;
3866
+ }
3867
+ export interface TrainingRuleDeleteResponse {
3868
+ deleted: boolean;
3869
+ rule_id: string;
3870
+ /**
3871
+ * The name this delete just spent. The delete is a soft delete and the name
3872
+ * index carries no partial predicate, so the row goes on holding the name
3873
+ * after it has left every list you can read, and no later rule in the
3874
+ * workspace can be called that. There is no purge and no undelete.
3875
+ */
3876
+ retained_name: string;
3877
+ cancelled_run_id: string | null;
3878
+ }
3879
+ export interface TrainingRunListResponse {
3880
+ runs: TrainingRunSummary[];
3881
+ total: number;
3882
+ }
3883
+ export interface TrainingRunResponse {
3884
+ run: TrainingRun;
3885
+ timeline: TrainingRunEvent[];
3886
+ /**
3887
+ * The standing benchmark's two absolute numbers for this run, beside the
3888
+ * paired verdict and never in place of it. Null when no benchmark is
3889
+ * attached to the rule.
3890
+ */
3891
+ benchmark: BenchmarkRun | null;
3892
+ links: TrainingRunLinks;
3893
+ available_actions: TrainingRunActions;
3894
+ }
3895
+ export interface TrainingRunActionResponse {
3896
+ run: TrainingRun;
3897
+ /** Set on rollback, the one action that changes what answers the traffic. */
3898
+ serving: TrainingRuleServing | null;
3899
+ }
3900
+ export interface EvaluationItemsResponse {
3901
+ items: EvaluationItem[];
3902
+ total: number;
3903
+ }
3904
+ export interface AgentSettingsResponse {
3905
+ settings: AgentSettings;
3906
+ }
3907
+ export interface InferenceAliasListResponse {
3908
+ aliases: InferenceAlias[];
3909
+ }
3910
+ export interface InferenceAliasResponse {
3911
+ alias: InferenceAlias;
3912
+ }
3913
+ export interface InferenceAliasDeleteResponse {
3914
+ deleted: boolean;
3915
+ }
3916
+ export interface BenchmarkListResponse {
3917
+ benchmarks: Benchmark[];
3918
+ total: number;
3919
+ }
3920
+ export interface BenchmarkItemsResponse {
3921
+ items: BenchmarkItem[];
3922
+ total: number;
3923
+ }
3924
+ /** The trend, newest first. Every replay is a point, including the scoreless ones. */
3925
+ export interface BenchmarkHistoryResponse {
3926
+ benchmark_id: string;
3927
+ points: BenchmarkHistoryPoint[];
3928
+ total: number;
3929
+ }