runbios-sdk 0.2.1-rc.98 → 0.2.2-dev.171
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +32 -2
- package/dist/client.d.ts +48 -1
- package/dist/client.js +64 -1
- package/dist/index.d.ts +15 -3
- package/dist/index.js +16 -2
- package/dist/resources/datasets.d.ts +8 -4
- package/dist/resources/datasets.js +15 -4
- package/dist/resources/gpu.js +7 -0
- package/dist/resources/inference.d.ts +33 -2
- package/dist/resources/inference.js +46 -1
- package/dist/resources/integrations.d.ts +21 -0
- package/dist/resources/integrations.js +20 -0
- package/dist/resources/loop.d.ts +850 -0
- package/dist/resources/loop.js +1189 -0
- package/dist/resources/training.d.ts +20 -8
- package/dist/resources/training.js +52 -9
- package/dist/types.d.ts +1980 -6
- package/dist/types.js +11 -1
- package/package.json +2 -2
package/dist/types.d.ts
CHANGED
|
@@ -8,13 +8,13 @@ export interface BiOSConfig {
|
|
|
8
8
|
orgId?: string;
|
|
9
9
|
/** Workspace ID. Optional when using API keys (resolved from the key). Can override for multi-workspace keys. */
|
|
10
10
|
workspaceId?: string;
|
|
11
|
-
/** Base URL for the API. Falls back to the RUNBIOS_BASE_URL environment variable (legacy: BIOS_BASE_URL), then the canonical https://api-
|
|
11
|
+
/** Base URL for the API. Falls back to the RUNBIOS_BASE_URL environment variable (legacy: BIOS_BASE_URL), then the canonical https://api-dev.runbios.ai hostname. */
|
|
12
12
|
baseUrl?: string;
|
|
13
13
|
/** Request timeout in milliseconds. Defaults to 30000. */
|
|
14
14
|
timeout?: number;
|
|
15
15
|
/** Default per-deployment inference key. Can be overridden per inference call. */
|
|
16
16
|
inferenceKey?: string;
|
|
17
|
-
/** Inference base URL. Defaults to baseUrl, then https://api-
|
|
17
|
+
/** Inference base URL. Defaults to baseUrl, then https://api-dev.runbios.ai. */
|
|
18
18
|
inferenceBaseUrl?: string;
|
|
19
19
|
/** End-to-end inference timeout in milliseconds. Defaults to 15 minutes. */
|
|
20
20
|
inferenceTimeout?: number;
|
|
@@ -396,8 +396,9 @@ export interface DatasetPreview {
|
|
|
396
396
|
}
|
|
397
397
|
/** Parameters for importing a dataset from HuggingFace Hub. */
|
|
398
398
|
export interface DatasetImportHFParams {
|
|
399
|
-
/** HuggingFace dataset repository ID (e.g. "
|
|
399
|
+
/** HuggingFace dataset repository ID (e.g. "HuggingFaceH4/ultrachat_200k"). */
|
|
400
400
|
repoId: string;
|
|
401
|
+
revision?: string;
|
|
401
402
|
/** Display name for the imported dataset. */
|
|
402
403
|
name?: string;
|
|
403
404
|
/** Dataset subset/config to import. */
|
|
@@ -418,6 +419,7 @@ export interface DatasetImportHFParams {
|
|
|
418
419
|
/** Parameters for registering a HuggingFace dataset without an integration. */
|
|
419
420
|
export interface DatasetRegisterHFParams {
|
|
420
421
|
repoId: string;
|
|
422
|
+
revision?: string;
|
|
421
423
|
name?: string;
|
|
422
424
|
workspaceId?: string;
|
|
423
425
|
description?: string;
|
|
@@ -441,6 +443,7 @@ export interface DatasetHubSearchParams {
|
|
|
441
443
|
export interface DatasetHubPreviewParams {
|
|
442
444
|
/** HuggingFace dataset ID. */
|
|
443
445
|
datasetId: string;
|
|
446
|
+
integrationId?: string;
|
|
444
447
|
/** Split to preview. Defaults to "train". */
|
|
445
448
|
split?: string;
|
|
446
449
|
/** Subset/config name. */
|
|
@@ -545,6 +548,21 @@ export interface TrainingCreateParams {
|
|
|
545
548
|
* large dataset without importing a trimmed copy.
|
|
546
549
|
*/
|
|
547
550
|
datasetSampleLimits?: Record<string, number>;
|
|
551
|
+
/**
|
|
552
|
+
* Per-dataset sampling (key = a dataset ID from datasetIds): exactly one of
|
|
553
|
+
* `rows` or `percent` (of the dataset's usable rows for this method). Asking
|
|
554
|
+
* for more than the dataset holds uses every usable row and the run says
|
|
555
|
+
* so. Resolved to a row count identically by preflight and create, recorded
|
|
556
|
+
* in the mix manifest, and reproduced exactly on resume.
|
|
557
|
+
*/
|
|
558
|
+
datasetSampling?: Record<string, DatasetSampling>;
|
|
559
|
+
/**
|
|
560
|
+
* How sampled rows are chosen: `first` (default) keeps the first N usable
|
|
561
|
+
* rows in file order -- what a curriculum wants; `random` draws a seeded
|
|
562
|
+
* uniform sample of the usable rows (kept in file order, seed = the mixing
|
|
563
|
+
* plan's seed, so a resume draws the same rows).
|
|
564
|
+
*/
|
|
565
|
+
datasetSamplingStrategy?: 'first' | 'random';
|
|
548
566
|
/** Training method. */
|
|
549
567
|
method: TrainingMethod;
|
|
550
568
|
/** Adapter type. */
|
|
@@ -565,8 +583,10 @@ export interface TrainingCreateParams {
|
|
|
565
583
|
name?: string;
|
|
566
584
|
/** Target workspace ID. */
|
|
567
585
|
workspaceId?: string;
|
|
568
|
-
/** Number of training epochs. */
|
|
586
|
+
/** Number of training epochs. Governs the full run when maxSteps is omitted. */
|
|
569
587
|
epochs?: number;
|
|
588
|
+
/** Optional hard step cap that overrides epochs. Omit for no cap and all configured epochs. */
|
|
589
|
+
maxSteps?: number;
|
|
570
590
|
/** Training batch size per device. */
|
|
571
591
|
batchSize?: number;
|
|
572
592
|
/** Gradient accumulation steps. */
|
|
@@ -609,7 +629,7 @@ export interface TrainingCreateParams {
|
|
|
609
629
|
integrationId?: string;
|
|
610
630
|
/** Existing network volume to attach. */
|
|
611
631
|
networkVolumeId?: string;
|
|
612
|
-
/**
|
|
632
|
+
/** @deprecated No effect. Prepared data is transient; exact resume rebuilds it from pinned source metadata and verifies its checksum. */
|
|
613
633
|
cacheDataset?: boolean;
|
|
614
634
|
/** Legacy dataset ordering mode. Prefer mixing for weighted/phased plans. */
|
|
615
635
|
datasetMixing?: 'shuffle' | 'sequential' | 'interleave' | 'random' | 'curriculum';
|
|
@@ -681,6 +701,13 @@ export interface TrainingJob {
|
|
|
681
701
|
status: TrainingJobStatus;
|
|
682
702
|
error_message?: string | null;
|
|
683
703
|
error_code?: string | null;
|
|
704
|
+
/**
|
|
705
|
+
* Who ended a `stopped` run: `user` (Stop button, API, queue cancel) or
|
|
706
|
+
* `platform` (wallet exhausted, account block). Empty when the run is not
|
|
707
|
+
* stopped, or was stopped before this was recorded. A `failed` run is a
|
|
708
|
+
* different fact and never carries a stop origin.
|
|
709
|
+
*/
|
|
710
|
+
stop_origin?: 'user' | 'platform' | '';
|
|
684
711
|
current_step?: number;
|
|
685
712
|
total_steps?: number;
|
|
686
713
|
current_loss?: number | null;
|
|
@@ -754,12 +781,37 @@ export interface TrainingListResponse {
|
|
|
754
781
|
limit: number;
|
|
755
782
|
offset: number;
|
|
756
783
|
}
|
|
784
|
+
export interface TrainingDeviceMetrics {
|
|
785
|
+
memory_used_mib: number | null;
|
|
786
|
+
memory_total_mib: number | null;
|
|
787
|
+
utilization_pct: number | null;
|
|
788
|
+
temperature_c: number | null;
|
|
789
|
+
}
|
|
790
|
+
export interface TrainingResourceMetrics {
|
|
791
|
+
scope: 'worker';
|
|
792
|
+
sampled_at: string | null;
|
|
793
|
+
received_at: string;
|
|
794
|
+
gpu_count: number | null;
|
|
795
|
+
gpu_memory_used_mib: number | null;
|
|
796
|
+
gpu_memory_total_mib: number | null;
|
|
797
|
+
gpu_utilization_pct: number | null;
|
|
798
|
+
gpu_temperature_c: number | null;
|
|
799
|
+
host_ram_used_gib: number | null;
|
|
800
|
+
host_ram_limit_gib: number | null;
|
|
801
|
+
host_ram_pct: number | null;
|
|
802
|
+
host_cpu_usage_pct: number | null;
|
|
803
|
+
host_cpu_limit_cores: number | null;
|
|
804
|
+
disk_used_gib: number | null;
|
|
805
|
+
disk_total_gib: number | null;
|
|
806
|
+
gpus: TrainingDeviceMetrics[];
|
|
807
|
+
}
|
|
757
808
|
/** Training metrics for a job. */
|
|
758
809
|
export interface TrainingMetrics {
|
|
759
810
|
metrics: MetricPoint[];
|
|
760
811
|
training_method?: string | null;
|
|
761
812
|
rlhf_type?: string | null;
|
|
762
813
|
graph_configs: MetricGraphConfig[];
|
|
814
|
+
resource_metrics?: TrainingResourceMetrics | null;
|
|
763
815
|
/** @deprecated Use metrics. Populated as a compatibility alias. */
|
|
764
816
|
steps?: MetricPoint[];
|
|
765
817
|
}
|
|
@@ -877,6 +929,8 @@ export interface TrainingPreflightDataset {
|
|
|
877
929
|
export interface TrainingPreflightWarning {
|
|
878
930
|
code: string;
|
|
879
931
|
message: string;
|
|
932
|
+
/** Request field the warning is about (e.g. `per_device_train_batch_size`), when there is one. */
|
|
933
|
+
field?: string;
|
|
880
934
|
}
|
|
881
935
|
/** Side-effect-free validation/sizing result; this endpoint never creates or bills a job. */
|
|
882
936
|
export interface TrainingPreflightResponse {
|
|
@@ -897,6 +951,131 @@ export interface TrainingPreflightResponse {
|
|
|
897
951
|
queue_eligible: boolean;
|
|
898
952
|
warnings: TrainingPreflightWarning[];
|
|
899
953
|
checked_at: string;
|
|
954
|
+
/**
|
|
955
|
+
* The trainer image's own sizing verdict for the requested GPU shape:
|
|
956
|
+
* recommended microbatch/accumulation/learning rate, predicted peak memory,
|
|
957
|
+
* minimum GPU count, wall-clock estimate, and whether the per-device batch
|
|
958
|
+
* you asked for is predicted to fit. Absent when no advisor is deployed.
|
|
959
|
+
*/
|
|
960
|
+
advisor?: TrainingAdvisorVerdict;
|
|
961
|
+
}
|
|
962
|
+
/** Parameters for `training.recommend()` (GET /api/training/recommend). */
|
|
963
|
+
export interface TrainingRecommendParams {
|
|
964
|
+
model: string;
|
|
965
|
+
modelRevision?: string;
|
|
966
|
+
integrationId?: string;
|
|
967
|
+
gpuType: string;
|
|
968
|
+
gpuCount?: number;
|
|
969
|
+
adapter?: 'full' | 'lora' | 'qlora';
|
|
970
|
+
method?: 'sft' | 'cpt' | 'pt';
|
|
971
|
+
maxLength?: number;
|
|
972
|
+
epochs?: number;
|
|
973
|
+
maxSteps?: number;
|
|
974
|
+
/** Your own microbatch, to be judged against the model. */
|
|
975
|
+
perDeviceTrainBatchSize?: number;
|
|
976
|
+
gradientAccumulationSteps?: number;
|
|
977
|
+
/** Datasets the job will train on; their measured token statistics feed the sizing. */
|
|
978
|
+
datasetIds?: string[];
|
|
979
|
+
workspaceId?: string;
|
|
980
|
+
}
|
|
981
|
+
/** Basis of one recommended value: measured on hardware, derived through a stated model, or an argued default. */
|
|
982
|
+
export type TrainingAdvisorBasis = 'measured' | 'derived' | 'judgement';
|
|
983
|
+
export interface TrainingAdvisorJustification {
|
|
984
|
+
field: string;
|
|
985
|
+
value: string;
|
|
986
|
+
reason: string;
|
|
987
|
+
basis: TrainingAdvisorBasis;
|
|
988
|
+
}
|
|
989
|
+
export interface TrainingAdvisorRecommendation {
|
|
990
|
+
per_device_train_batch_size: number;
|
|
991
|
+
gradient_accumulation_steps: number;
|
|
992
|
+
global_batch_size: number;
|
|
993
|
+
activation_checkpoint: string;
|
|
994
|
+
compile: boolean;
|
|
995
|
+
learning_rate: number;
|
|
996
|
+
warmup_steps: number;
|
|
997
|
+
parallelism: {
|
|
998
|
+
dp_replicate: number;
|
|
999
|
+
dp_shard: number;
|
|
1000
|
+
tp: number;
|
|
1001
|
+
pp: number;
|
|
1002
|
+
cp: number;
|
|
1003
|
+
ep: number;
|
|
1004
|
+
};
|
|
1005
|
+
/** Predicted peak reserved GPU memory per device, GiB. */
|
|
1006
|
+
predicted_peak_gb: number;
|
|
1007
|
+
/** (median, worst) percent the prediction ran over measurement on the calibration rows. */
|
|
1008
|
+
memory_band_percent: [number, number];
|
|
1009
|
+
memory_class: string;
|
|
1010
|
+
predicted_mfu: number;
|
|
1011
|
+
supervised_tokens_per_step: number;
|
|
1012
|
+
predicted_roughness: number;
|
|
1013
|
+
tokens_per_second: number;
|
|
1014
|
+
throughput_basis: 'measured' | 'derived';
|
|
1015
|
+
wall_clock: {
|
|
1016
|
+
steps_per_epoch: number;
|
|
1017
|
+
total_steps: number;
|
|
1018
|
+
training_hours: number;
|
|
1019
|
+
startup_minutes_estimate: number;
|
|
1020
|
+
basis: TrainingAdvisorBasis;
|
|
1021
|
+
} | null;
|
|
1022
|
+
}
|
|
1023
|
+
export interface TrainingAdvisorUserShape {
|
|
1024
|
+
per_device_train_batch_size: number;
|
|
1025
|
+
fits: boolean;
|
|
1026
|
+
largest_fitting_batch?: number;
|
|
1027
|
+
predicted_peak_gb?: number;
|
|
1028
|
+
reason?: string;
|
|
1029
|
+
/** Same global batch, a microbatch that fits. Present only when `fits` is false. */
|
|
1030
|
+
suggested?: {
|
|
1031
|
+
per_device_train_batch_size: number;
|
|
1032
|
+
gradient_accumulation_steps: number;
|
|
1033
|
+
reason: string;
|
|
1034
|
+
};
|
|
1035
|
+
}
|
|
1036
|
+
/**
|
|
1037
|
+
* Verdict of the training advisor. `available: false` means no advisor is
|
|
1038
|
+
* deployed or it did not answer; nothing else is populated then.
|
|
1039
|
+
*/
|
|
1040
|
+
export interface TrainingAdvisorVerdict {
|
|
1041
|
+
available: boolean;
|
|
1042
|
+
reason?: string;
|
|
1043
|
+
fits?: boolean;
|
|
1044
|
+
/** Smallest GPU count of this type the job fits on; null when none up to the per-job cap. */
|
|
1045
|
+
min_gpu_count?: number | null;
|
|
1046
|
+
gpu?: {
|
|
1047
|
+
platform_type: string;
|
|
1048
|
+
sized_as: string;
|
|
1049
|
+
capacity_gb: number;
|
|
1050
|
+
capacity_measured: boolean;
|
|
1051
|
+
};
|
|
1052
|
+
model?: {
|
|
1053
|
+
params_total_b: number;
|
|
1054
|
+
params_active_b: number;
|
|
1055
|
+
is_moe: boolean;
|
|
1056
|
+
is_vlm: boolean;
|
|
1057
|
+
has_linear_attention: boolean;
|
|
1058
|
+
};
|
|
1059
|
+
dataset_measured?: boolean;
|
|
1060
|
+
seq_len?: number;
|
|
1061
|
+
gpu_count?: number;
|
|
1062
|
+
recommended?: TrainingAdvisorRecommendation;
|
|
1063
|
+
user_shape?: TrainingAdvisorUserShape;
|
|
1064
|
+
justifications?: TrainingAdvisorJustification[];
|
|
1065
|
+
warnings?: string[];
|
|
1066
|
+
dataset_stats_used?: {
|
|
1067
|
+
num_rows: number;
|
|
1068
|
+
avg_tokens_per_sample?: number;
|
|
1069
|
+
avg_supervised_tokens_per_sample?: number;
|
|
1070
|
+
has_images: boolean;
|
|
1071
|
+
estimated: boolean;
|
|
1072
|
+
};
|
|
1073
|
+
measured_history?: {
|
|
1074
|
+
tokens_per_second: number;
|
|
1075
|
+
memory_anchors: number;
|
|
1076
|
+
};
|
|
1077
|
+
model_id?: string;
|
|
1078
|
+
model_revision?: string;
|
|
900
1079
|
}
|
|
901
1080
|
/** One method, algorithm, or adapter reported by the pinned training engine. */
|
|
902
1081
|
export interface TrainingCapabilityChoice {
|
|
@@ -916,6 +1095,8 @@ export interface TrainingConfigFieldCapability {
|
|
|
916
1095
|
label: string;
|
|
917
1096
|
type: 'integer' | 'number' | 'boolean' | 'string' | 'string_array' | 'string_or_string_array' | 'object';
|
|
918
1097
|
default: unknown;
|
|
1098
|
+
default_by_adapter?: Record<string, number>;
|
|
1099
|
+
requires_step_horizon?: boolean;
|
|
919
1100
|
enabled: boolean;
|
|
920
1101
|
disabled_reason?: string;
|
|
921
1102
|
aliases?: string[];
|
|
@@ -1052,6 +1233,12 @@ export interface GPUPricingResponse {
|
|
|
1052
1233
|
stale?: boolean;
|
|
1053
1234
|
stale_message?: string;
|
|
1054
1235
|
}
|
|
1236
|
+
/** One dataset's sampling request: exactly one of rows or percent. */
|
|
1237
|
+
export interface DatasetSampling {
|
|
1238
|
+
rows?: number;
|
|
1239
|
+
/** Percent (0-100) of the dataset's usable rows for the training method. */
|
|
1240
|
+
percent?: number;
|
|
1241
|
+
}
|
|
1055
1242
|
/** Parameters understood by the authenticated training GPU-options endpoint. */
|
|
1056
1243
|
export interface GPUOptionsParams {
|
|
1057
1244
|
modelId: string;
|
|
@@ -1067,6 +1254,17 @@ export interface GPUOptionsParams {
|
|
|
1067
1254
|
rlhfType?: RLHFAlgorithm | string;
|
|
1068
1255
|
modelParamsB?: number;
|
|
1069
1256
|
modelActiveParamsB?: number;
|
|
1257
|
+
/** Training sequence length the sizing should assume (tokens). */
|
|
1258
|
+
maxLength?: number;
|
|
1259
|
+
/**
|
|
1260
|
+
* Effective batch (samples per optimizer step) -- the quality decision made
|
|
1261
|
+
* before a GPU is chosen. Each option then answers with the per-device split
|
|
1262
|
+
* that runs it there (`recommended_micro_batch` x `recommended_grad_accum`
|
|
1263
|
+
* x `required_count`), and the minimum GPU count is sized at micro-batch 1.
|
|
1264
|
+
*/
|
|
1265
|
+
effectiveBatch?: number;
|
|
1266
|
+
/** Per-device batch to size as typed instead (ignored when effectiveBatch is set). */
|
|
1267
|
+
perDeviceTrainBatchSize?: number;
|
|
1070
1268
|
}
|
|
1071
1269
|
/** One model-aware GPU option with live stock and total-price context. */
|
|
1072
1270
|
export interface GPUOption {
|
|
@@ -1087,6 +1285,10 @@ export interface GPUOption {
|
|
|
1087
1285
|
bookable: boolean;
|
|
1088
1286
|
reason?: string;
|
|
1089
1287
|
checked_at?: string;
|
|
1288
|
+
/** Present when the request stated an effective batch: its split on this card at required_count. */
|
|
1289
|
+
effective_batch?: number;
|
|
1290
|
+
recommended_micro_batch?: number;
|
|
1291
|
+
recommended_grad_accum?: number;
|
|
1090
1292
|
}
|
|
1091
1293
|
/** Actionable alternative returned when no GPU option is currently bookable. */
|
|
1092
1294
|
export interface GPUOptionSuggestion {
|
|
@@ -1116,6 +1318,10 @@ export interface GPUOptionsResponse {
|
|
|
1116
1318
|
storage_gb: number;
|
|
1117
1319
|
price_per_hour_cents: number;
|
|
1118
1320
|
total_price_per_hour_cents: number;
|
|
1321
|
+
/** Present when the request stated an effective batch. */
|
|
1322
|
+
effective_batch?: number;
|
|
1323
|
+
per_device_train_batch_size?: number;
|
|
1324
|
+
gradient_accumulation_steps?: number;
|
|
1119
1325
|
};
|
|
1120
1326
|
suggestions?: GPUOptionSuggestion[];
|
|
1121
1327
|
}
|
|
@@ -1817,7 +2023,7 @@ export interface ApiKey {
|
|
|
1817
2023
|
* on purpose: a scope added to the catalog must not make an existing SDK build
|
|
1818
2024
|
* reject a key it just read back from the API.
|
|
1819
2025
|
*/
|
|
1820
|
-
export type ApiKeyScope = 'models:read' | 'datasets:read' | 'datasets:write' | 'training:read' | 'training:write' | 'billing:read' | 'deployments:read' | 'deployments:write' | 'analytics:read' | 'integrations:read' | 'integrations:write' | 'skills:read' | 'skills:write' | 'serverless' | (string & {});
|
|
2026
|
+
export type ApiKeyScope = 'models:read' | 'datasets:read' | 'datasets:write' | 'training:read' | 'training:write' | 'billing:read' | 'deployments:read' | 'deployments:write' | 'analytics:read' | 'integrations:read' | 'integrations:write' | 'skills:read' | 'skills:write' | 'loop:read' | 'loop:write' | 'serverless' | (string & {});
|
|
1821
2027
|
/** Result of introspecting an API key — shows what it can do. */
|
|
1822
2028
|
export interface ApiKeyIntrospection {
|
|
1823
2029
|
auth_type: 'api_key' | 'jwt';
|
|
@@ -1953,3 +2159,1771 @@ export interface ChatCompletionUsage {
|
|
|
1953
2159
|
total_tokens: number;
|
|
1954
2160
|
prompt_tokens_details?: PromptTokensDetails;
|
|
1955
2161
|
}
|
|
2162
|
+
/** What a training set is shaped for. The three need different things. */
|
|
2163
|
+
export type LoopMethod = 'sft' | 'dpo' | 'grpo' | 'kto';
|
|
2164
|
+
/** Who judged an answer. */
|
|
2165
|
+
export type LoopSource = 'human' | 'verifier' | 'judge' | 'behavioural';
|
|
2166
|
+
/** The judgement itself. */
|
|
2167
|
+
/**
|
|
2168
|
+
* The judgement recorded on an answer.
|
|
2169
|
+
*
|
|
2170
|
+
* `accepted` / `rejected` are the thumbs (see `upvote` / `downvote`).
|
|
2171
|
+
* `edited` says the model was WRONG and carries the better answer.
|
|
2172
|
+
* `gold` records the reference answer for the question, making no claim about
|
|
2173
|
+
* whether the model was right — which is why it is separate from `edited`.
|
|
2174
|
+
*/
|
|
2175
|
+
export type LoopVerdict = 'accepted' | 'rejected' | 'edited' | 'scored' | 'gold';
|
|
2176
|
+
/** A function an assistant turn asked to invoke. `arguments` is JSON-encoded. */
|
|
2177
|
+
export interface LoopToolCall {
|
|
2178
|
+
id?: string;
|
|
2179
|
+
type?: string;
|
|
2180
|
+
function: {
|
|
2181
|
+
name: string;
|
|
2182
|
+
arguments: string;
|
|
2183
|
+
};
|
|
2184
|
+
}
|
|
2185
|
+
/** One conversational turn, in the chat-completions shape. */
|
|
2186
|
+
export interface LoopMessage {
|
|
2187
|
+
role: string;
|
|
2188
|
+
content?: string;
|
|
2189
|
+
tool_calls?: LoopToolCall[];
|
|
2190
|
+
tool_call_id?: string;
|
|
2191
|
+
name?: string;
|
|
2192
|
+
}
|
|
2193
|
+
export interface LoopCaptureParams {
|
|
2194
|
+
/** The source being recorded. Must be enabled first, or nothing is stored. */
|
|
2195
|
+
deployment_id: string;
|
|
2196
|
+
model: string;
|
|
2197
|
+
messages: LoopMessage[];
|
|
2198
|
+
completion?: string;
|
|
2199
|
+
/** What the answering turn invoked, if anything. */
|
|
2200
|
+
tool_calls?: LoopToolCall[];
|
|
2201
|
+
/** The tool schema the model was offered. Without it, tool training invents names. */
|
|
2202
|
+
tools?: unknown;
|
|
2203
|
+
model_version?: string;
|
|
2204
|
+
/** Ties multi-turn work together. */
|
|
2205
|
+
conversation_id?: string;
|
|
2206
|
+
/** Your own idempotency handle. Replaying it returns the same trace. */
|
|
2207
|
+
request_id?: string;
|
|
2208
|
+
prompt_tokens?: number;
|
|
2209
|
+
completion_tokens?: number;
|
|
2210
|
+
latency_ms?: number;
|
|
2211
|
+
metadata?: Record<string, unknown>;
|
|
2212
|
+
}
|
|
2213
|
+
export interface LoopCaptureResult {
|
|
2214
|
+
captured: boolean;
|
|
2215
|
+
/** Present when captured. Ours, never the id you sent. */
|
|
2216
|
+
trace_id?: string;
|
|
2217
|
+
/** Set when something was removed before the record was written. */
|
|
2218
|
+
redacted?: boolean;
|
|
2219
|
+
download_url?: string;
|
|
2220
|
+
/** Why nothing was stored, e.g. `capture_disabled`, `not_sampled`. */
|
|
2221
|
+
reason?: string;
|
|
2222
|
+
}
|
|
2223
|
+
export interface LoopImportParams {
|
|
2224
|
+
/**
|
|
2225
|
+
* Where this came from. Required, and becomes the id every imported
|
|
2226
|
+
* conversation is filed under, so the import stays sliceable later.
|
|
2227
|
+
*/
|
|
2228
|
+
source: string;
|
|
2229
|
+
/** The file, already parsed. At most 5000 rows per call. */
|
|
2230
|
+
rows: Record<string, unknown>[];
|
|
2231
|
+
/** Which model produced these, when the rows do not say per-row. */
|
|
2232
|
+
model?: string;
|
|
2233
|
+
/** Applied to every row: importing a dump is when somebody knows what it is. */
|
|
2234
|
+
labels?: string[];
|
|
2235
|
+
attributes?: Record<string, string>;
|
|
2236
|
+
/**
|
|
2237
|
+
* Repeat the `import_id` a previous call reported to make this call a RETRY
|
|
2238
|
+
* of that one: rows that already arrived come back under `already_present`
|
|
2239
|
+
* instead of being stored again.
|
|
2240
|
+
*
|
|
2241
|
+
* Leave it out and this call is its own import. The endpoint will NOT guess:
|
|
2242
|
+
* two calls carrying the same rows are as likely to be two pages of one
|
|
2243
|
+
* export as one call sent twice, and guessing "retry" silently drops the
|
|
2244
|
+
* second copy of every conversation a file lists more than once. Every
|
|
2245
|
+
* response carries the token it was filed under, so a retry is always
|
|
2246
|
+
* available and never has to be inferred.
|
|
2247
|
+
*
|
|
2248
|
+
* SEND THE ROWS AS YOU SENT THEM. Without `row_ids` a row is matched on its
|
|
2249
|
+
* text, and the match survives re-ordered keys and different whitespace but
|
|
2250
|
+
* NOT a number that has been re-spelled: `100`, `1e2` and `100.0` are three
|
|
2251
|
+
* different rows. A round trip through most JSON libraries re-spells numbers
|
|
2252
|
+
* (Python turns `1e2` into `100.0`), so a retry built by re-serialising a
|
|
2253
|
+
* parsed file can store rows a second time. Keep the bytes you sent, or send
|
|
2254
|
+
* `row_ids` and stop depending on the text at all.
|
|
2255
|
+
*/
|
|
2256
|
+
import_id?: string;
|
|
2257
|
+
/**
|
|
2258
|
+
* Where this page starts in the file, counting from 0. Send it when you split
|
|
2259
|
+
* one import across several calls under one `import_id`, so two pages are not
|
|
2260
|
+
* matched against each other row for row. Ignored when you send `row_ids`.
|
|
2261
|
+
*/
|
|
2262
|
+
row_offset?: number;
|
|
2263
|
+
/**
|
|
2264
|
+
* Your own id for each row, in the same order as `rows`: a ticket number, a
|
|
2265
|
+
* conversation id, whatever the export already carries.
|
|
2266
|
+
*
|
|
2267
|
+
* The strongest form of identity, and the one to reach for with a file big
|
|
2268
|
+
* enough to page. It SUPERSEDES `import_id` and `row_offset`: with it,
|
|
2269
|
+
* chunking, ordering and subsets all stop mattering and any part of the file
|
|
2270
|
+
* can be re-sent exactly. One per row or none at all, and no two rows in a
|
|
2271
|
+
* call may share an id: both are refused rather than silently merging two
|
|
2272
|
+
* conversations into one.
|
|
2273
|
+
*
|
|
2274
|
+
* THE ID IS THE IDENTITY, AND THE ROW'S TEXT IS NOT PART OF IT. A row sent
|
|
2275
|
+
* again under an id this source already imported comes back under
|
|
2276
|
+
* `already_present` and the stored conversation is left as it was, even if
|
|
2277
|
+
* you changed its text. A CORRECTION DOES NOT LAND THIS WAY: send it under a
|
|
2278
|
+
* new id. That is the trade for making every retry, page and subset safe -
|
|
2279
|
+
* the text is never compared, so nothing your serialiser does to it can turn
|
|
2280
|
+
* a retry into a second import.
|
|
2281
|
+
*/
|
|
2282
|
+
row_ids?: string[];
|
|
2283
|
+
}
|
|
2284
|
+
/**
|
|
2285
|
+
* What an import did.
|
|
2286
|
+
*
|
|
2287
|
+
* A call where every row stored resolves with this. A call where SOME rows
|
|
2288
|
+
* could not be stored rejects with {@link LoopImportIncompleteError}, whose
|
|
2289
|
+
* `outcome` is this same shape: the counts are the whole answer either way, so
|
|
2290
|
+
* read them from the error exactly as you would from a success.
|
|
2291
|
+
*/
|
|
2292
|
+
export interface LoopImportResult {
|
|
2293
|
+
/**
|
|
2294
|
+
* The token this call was filed under, whether you sent one or it was minted
|
|
2295
|
+
* for you. Send the same rows again with this `import_id` to retry the call.
|
|
2296
|
+
* Present on every response, success or failure, because a call can succeed
|
|
2297
|
+
* and still need retrying when the answer never reached you.
|
|
2298
|
+
*/
|
|
2299
|
+
import_id?: string;
|
|
2300
|
+
/** How many conversations this call created. */
|
|
2301
|
+
imported: number;
|
|
2302
|
+
/**
|
|
2303
|
+
* How many rows an earlier call under the SAME `import_id` (or carrying the
|
|
2304
|
+
* same `row_ids`) had already imported, and so were not stored a second time.
|
|
2305
|
+
* Zero on a first import, and zero on any call that repeated neither: a call
|
|
2306
|
+
* that does not say it is a retry is its own import, which is what lets a
|
|
2307
|
+
* file sent in pages land complete.
|
|
2308
|
+
*/
|
|
2309
|
+
already_present?: number;
|
|
2310
|
+
/**
|
|
2311
|
+
* How many verdicts THIS CALL WROTE. Reported apart from `imported` because
|
|
2312
|
+
* it is the difference between data you can train on and data somebody still
|
|
2313
|
+
* has to look at.
|
|
2314
|
+
*
|
|
2315
|
+
* Re-sending a row whose verdict already landed writes nothing and counts
|
|
2316
|
+
* nothing, so a plain retry reads 0. Re-sending one that came back in
|
|
2317
|
+
* `verdicts_not_saved` DOES count here when the write succeeds this time,
|
|
2318
|
+
* even though the row itself is counted under `already_present`: that is the
|
|
2319
|
+
* first time anybody recorded that verdict, not a second reviewer. This is
|
|
2320
|
+
* the field to read to confirm a repair landed.
|
|
2321
|
+
*/
|
|
2322
|
+
reviewed: number;
|
|
2323
|
+
needs_review: number;
|
|
2324
|
+
/**
|
|
2325
|
+
* How many rows could not be READ as a conversation. The file's shape is
|
|
2326
|
+
* wrong, and changing the file is what fixes it.
|
|
2327
|
+
*/
|
|
2328
|
+
refused: number;
|
|
2329
|
+
/**
|
|
2330
|
+
* How many rows were understood and then could not be stored. Counted apart
|
|
2331
|
+
* from `refused`, because the two are different claims and rewriting a file
|
|
2332
|
+
* that was already correct fixes nothing. A call with any of these rejects
|
|
2333
|
+
* rather than resolving; read it off {@link LoopImportIncompleteError.outcome}.
|
|
2334
|
+
*/
|
|
2335
|
+
not_saved?: number;
|
|
2336
|
+
/**
|
|
2337
|
+
* WHICH rows those were, counting from 1 in the order you sent them, so you
|
|
2338
|
+
* can say what is missing rather than only how much.
|
|
2339
|
+
*
|
|
2340
|
+
* NOT A SMALLER FILE TO SEND BACK unless you sent `row_ids`. Without them a
|
|
2341
|
+
* row is identified by its place in the call, so a shorter list moves every
|
|
2342
|
+
* row after the gap and each one is stored again. Re-send the whole set of
|
|
2343
|
+
* rows, in the same order, with the same `import_id`; what already arrived
|
|
2344
|
+
* comes back under `already_present`. With `row_ids` a row carries its own
|
|
2345
|
+
* identity and any subset is exact.
|
|
2346
|
+
*/
|
|
2347
|
+
not_saved_rows?: number[];
|
|
2348
|
+
/**
|
|
2349
|
+
* Rows whose CONVERSATION stored and whose VERDICT did not, and which rows
|
|
2350
|
+
* those were. A preference pair or a thumbs label is two writes, and the
|
|
2351
|
+
* second can fail on its own: the conversation is then in the loop carrying
|
|
2352
|
+
* nobody's judgement, counted under `needs_review` rather than `reviewed`,
|
|
2353
|
+
* and not trainable. The call rejects with {@link LoopImportIncompleteError},
|
|
2354
|
+
* and sending the rows again under the same `import_id` records the verdict
|
|
2355
|
+
* without storing the conversation twice.
|
|
2356
|
+
*/
|
|
2357
|
+
verdicts_not_saved?: number;
|
|
2358
|
+
verdicts_not_saved_rows?: number[];
|
|
2359
|
+
/**
|
|
2360
|
+
* How many conversations of each shape THIS CALL created. Rows that were
|
|
2361
|
+
* refused, rows that failed to store and rows an earlier import already had
|
|
2362
|
+
* are not in here: the counts and the sentences in `notes` describe the same
|
|
2363
|
+
* call, so neither can claim a conversation is waiting for review when none
|
|
2364
|
+
* was written.
|
|
2365
|
+
*/
|
|
2366
|
+
by_shape: Record<string, number>;
|
|
2367
|
+
refused_why: Record<string, number>;
|
|
2368
|
+
/**
|
|
2369
|
+
* The conversations from this file that are now in the loop, in file order:
|
|
2370
|
+
* the ones this call created and the ones it found already there. Without
|
|
2371
|
+
* them a caller whose import half-failed has a number and no way to reach
|
|
2372
|
+
* what landed: no way to label it, review it, or delete it and start again.
|
|
2373
|
+
*/
|
|
2374
|
+
trace_ids?: string[];
|
|
2375
|
+
notes: string[];
|
|
2376
|
+
}
|
|
2377
|
+
export interface LoopSignal {
|
|
2378
|
+
id: string;
|
|
2379
|
+
trace_id: string;
|
|
2380
|
+
source: LoopSource;
|
|
2381
|
+
verdict: LoopVerdict;
|
|
2382
|
+
correction: string | null;
|
|
2383
|
+
score: number | null;
|
|
2384
|
+
ground_truth: string | null;
|
|
2385
|
+
reason: string | null;
|
|
2386
|
+
author: string | null;
|
|
2387
|
+
created_at: string;
|
|
2388
|
+
}
|
|
2389
|
+
export interface LoopSignalParams {
|
|
2390
|
+
verdict: LoopVerdict;
|
|
2391
|
+
source?: LoopSource;
|
|
2392
|
+
/** Required when verdict is `edited`. The answer the model should have given. */
|
|
2393
|
+
correction?: string;
|
|
2394
|
+
/** Required when verdict is `scored`. */
|
|
2395
|
+
score?: number;
|
|
2396
|
+
/** A value or fact the answer can be checked against. GRPO needs one. */
|
|
2397
|
+
ground_truth?: string;
|
|
2398
|
+
reason?: string;
|
|
2399
|
+
author?: string;
|
|
2400
|
+
metadata?: Record<string, unknown>;
|
|
2401
|
+
}
|
|
2402
|
+
export interface LoopTrace {
|
|
2403
|
+
id: string;
|
|
2404
|
+
workspace_id: string;
|
|
2405
|
+
deployment_id: string | null;
|
|
2406
|
+
model: string;
|
|
2407
|
+
model_version: string | null;
|
|
2408
|
+
messages: LoopMessage[];
|
|
2409
|
+
completion: string;
|
|
2410
|
+
tool_calls?: LoopToolCall[];
|
|
2411
|
+
tools?: unknown;
|
|
2412
|
+
conversation_id: string | null;
|
|
2413
|
+
request_id: string | null;
|
|
2414
|
+
prompt_tokens: number | null;
|
|
2415
|
+
completion_tokens: number | null;
|
|
2416
|
+
latency_ms: number | null;
|
|
2417
|
+
redacted_at: string | null;
|
|
2418
|
+
/** What the redaction pass removed, counted by rule. */
|
|
2419
|
+
redaction_report?: Record<string, number>;
|
|
2420
|
+
created_at: string;
|
|
2421
|
+
expires_at: string;
|
|
2422
|
+
signals?: LoopSignal[];
|
|
2423
|
+
}
|
|
2424
|
+
export interface LoopTraceListParams {
|
|
2425
|
+
deployment_id?: string;
|
|
2426
|
+
conversation_id?: string;
|
|
2427
|
+
from?: string;
|
|
2428
|
+
to?: string;
|
|
2429
|
+
/** Only conversations that already carry a verdict. */
|
|
2430
|
+
signalled?: boolean;
|
|
2431
|
+
/** One bare tag. A tag with children matches them too. */
|
|
2432
|
+
label?: string;
|
|
2433
|
+
/**
|
|
2434
|
+
* Named dimensions, ANDed together: `{ category: 'billing', language: 'es' }`.
|
|
2435
|
+
* Exact within their key — `source=support` does not match `team=support`.
|
|
2436
|
+
*/
|
|
2437
|
+
attributes?: Record<string, string>;
|
|
2438
|
+
/** Only the conversations nobody has described yet. */
|
|
2439
|
+
unlabelled?: boolean;
|
|
2440
|
+
limit?: number;
|
|
2441
|
+
offset?: number;
|
|
2442
|
+
/** Only what ONE model answered — the filter distillation is made of. */
|
|
2443
|
+
model?: string;
|
|
2444
|
+
/** `captured` (our model's own behaviour) or `imported` (a file somebody brought). */
|
|
2445
|
+
origin?: 'captured' | 'imported';
|
|
2446
|
+
}
|
|
2447
|
+
export interface LoopTraceListResponse {
|
|
2448
|
+
traces: LoopTrace[];
|
|
2449
|
+
total: number;
|
|
2450
|
+
limit: number;
|
|
2451
|
+
offset: number;
|
|
2452
|
+
}
|
|
2453
|
+
export interface LoopDatasetCreateParams {
|
|
2454
|
+
name: string;
|
|
2455
|
+
method: LoopMethod;
|
|
2456
|
+
deployment_id?: string;
|
|
2457
|
+
from?: string;
|
|
2458
|
+
to?: string;
|
|
2459
|
+
/** Narrow to one slice. A label with children selects them too. */
|
|
2460
|
+
label?: string;
|
|
2461
|
+
/**
|
|
2462
|
+
* Take this many of what the filters matched. Which ones is arbitrary but
|
|
2463
|
+
* REPEATABLE: the same conversations always give the same slice, so two
|
|
2464
|
+
* builds of one spec describe the same set.
|
|
2465
|
+
*/
|
|
2466
|
+
sample?: number;
|
|
2467
|
+
/** Holds back your most recent work, not a random slice. 0-50. */
|
|
2468
|
+
holdout_percent?: number;
|
|
2469
|
+
max_items?: number;
|
|
2470
|
+
/**
|
|
2471
|
+
* Named dimensions, ANDed with each other and with `label`:
|
|
2472
|
+
* `{ category: 'billing', language: 'es' }`. This is what turns one captured
|
|
2473
|
+
* corpus into a different dataset for every task somebody trains for.
|
|
2474
|
+
*/
|
|
2475
|
+
attributes?: Record<string, string>;
|
|
2476
|
+
/** Only the conversations nobody has described yet. */
|
|
2477
|
+
unlabelled?: boolean;
|
|
2478
|
+
/** Only what ONE model answered — the filter distillation is made of. */
|
|
2479
|
+
model?: string;
|
|
2480
|
+
/** `captured` (our model's own behaviour) or `imported` (a file somebody brought). */
|
|
2481
|
+
origin?: 'captured' | 'imported';
|
|
2482
|
+
}
|
|
2483
|
+
export interface LoopDataset {
|
|
2484
|
+
id: string;
|
|
2485
|
+
workspace_id: string;
|
|
2486
|
+
name: string;
|
|
2487
|
+
method: LoopMethod;
|
|
2488
|
+
status: string;
|
|
2489
|
+
spec: Record<string, unknown>;
|
|
2490
|
+
item_count: number;
|
|
2491
|
+
considered_count: number;
|
|
2492
|
+
/** Why rows were left out, by reason. */
|
|
2493
|
+
rejected_counts: Record<string, number>;
|
|
2494
|
+
holdout_count: number;
|
|
2495
|
+
holdout_cutoff: string | null;
|
|
2496
|
+
created_by: string | null;
|
|
2497
|
+
created_at: string;
|
|
2498
|
+
completed_at: string | null;
|
|
2499
|
+
download_url: string;
|
|
2500
|
+
}
|
|
2501
|
+
export interface LoopDatasetListParams {
|
|
2502
|
+
method?: LoopMethod;
|
|
2503
|
+
limit?: number;
|
|
2504
|
+
offset?: number;
|
|
2505
|
+
}
|
|
2506
|
+
export interface LoopDatasetItem {
|
|
2507
|
+
id: number;
|
|
2508
|
+
trace_id: string;
|
|
2509
|
+
split: 'train' | 'holdout';
|
|
2510
|
+
payload: Record<string, unknown>;
|
|
2511
|
+
dedup_key: string;
|
|
2512
|
+
created_at: string;
|
|
2513
|
+
/** The conversation this row was built from. */
|
|
2514
|
+
trace_url: string;
|
|
2515
|
+
}
|
|
2516
|
+
export interface LoopConfig {
|
|
2517
|
+
deployment_id: string;
|
|
2518
|
+
workspace_id: string;
|
|
2519
|
+
enabled: boolean;
|
|
2520
|
+
retention_days: number;
|
|
2521
|
+
sample_rate: number;
|
|
2522
|
+
enabled_by: string | null;
|
|
2523
|
+
enabled_at: string | null;
|
|
2524
|
+
updated_at: string;
|
|
2525
|
+
auto_grade: boolean;
|
|
2526
|
+
}
|
|
2527
|
+
export interface LoopBuildRuleParams {
|
|
2528
|
+
name: string;
|
|
2529
|
+
method: string;
|
|
2530
|
+
/**
|
|
2531
|
+
* How many rows reviewed SINCE THE LAST BUILD must exist before this fires
|
|
2532
|
+
* again. Minimum 10. Counting the whole corpus instead would fire the rule
|
|
2533
|
+
* every interval forever, because a total that has crossed a threshold stays
|
|
2534
|
+
* across it.
|
|
2535
|
+
*/
|
|
2536
|
+
min_new_rows?: number;
|
|
2537
|
+
enabled?: boolean;
|
|
2538
|
+
/** The same selection `createDataset` takes, replayed verbatim. */
|
|
2539
|
+
spec?: LoopDatasetCreateParams;
|
|
2540
|
+
}
|
|
2541
|
+
export interface LoopBuildRule {
|
|
2542
|
+
id: string;
|
|
2543
|
+
workspace_id: string;
|
|
2544
|
+
name: string;
|
|
2545
|
+
method: string;
|
|
2546
|
+
spec: Record<string, unknown>;
|
|
2547
|
+
min_new_rows: number;
|
|
2548
|
+
enabled: boolean;
|
|
2549
|
+
last_built_at: string | null;
|
|
2550
|
+
last_dataset_id: string | null;
|
|
2551
|
+
/**
|
|
2552
|
+
* Why the rule did not fire last time it was checked. A rule quiet because
|
|
2553
|
+
* it is waiting looks exactly like one quiet because it is broken; this is
|
|
2554
|
+
* the difference.
|
|
2555
|
+
*/
|
|
2556
|
+
last_reason: string | null;
|
|
2557
|
+
last_checked_at: string | null;
|
|
2558
|
+
created_by: string | null;
|
|
2559
|
+
created_at: string;
|
|
2560
|
+
updated_at: string;
|
|
2561
|
+
}
|
|
2562
|
+
export interface LoopConfigParams {
|
|
2563
|
+
enabled: boolean;
|
|
2564
|
+
retention_days?: number;
|
|
2565
|
+
/** Deterministic per conversation, so turns are never split. 0 < rate <= 1. */
|
|
2566
|
+
sample_rate?: number;
|
|
2567
|
+
/**
|
|
2568
|
+
* Continuous rule-based scoring for this source. OMITTED MEANS UNCHANGED:
|
|
2569
|
+
* a caller saving retention must not switch grading off for a workspace
|
|
2570
|
+
* that turned it on.
|
|
2571
|
+
*/
|
|
2572
|
+
auto_grade?: boolean;
|
|
2573
|
+
}
|
|
2574
|
+
/** One thing a judge scores, separately from the others. */
|
|
2575
|
+
export interface LoopJudgeDimension {
|
|
2576
|
+
key: string;
|
|
2577
|
+
description?: string;
|
|
2578
|
+
}
|
|
2579
|
+
/** Which slice of the corpus a judge is responsible for. */
|
|
2580
|
+
export interface LoopJudgeSelection {
|
|
2581
|
+
label?: string;
|
|
2582
|
+
deployment_id?: string;
|
|
2583
|
+
from?: string;
|
|
2584
|
+
to?: string;
|
|
2585
|
+
sample?: number;
|
|
2586
|
+
/** Skip conversations that already carry a judge verdict. */
|
|
2587
|
+
only_unscored?: boolean;
|
|
2588
|
+
}
|
|
2589
|
+
export interface LoopJudge {
|
|
2590
|
+
id: string;
|
|
2591
|
+
workspace_id: string;
|
|
2592
|
+
name: string;
|
|
2593
|
+
instructions: string;
|
|
2594
|
+
dimensions: LoopJudgeDimension[];
|
|
2595
|
+
selection: LoopJudgeSelection;
|
|
2596
|
+
model: string | null;
|
|
2597
|
+
write_gold: boolean;
|
|
2598
|
+
enabled: boolean;
|
|
2599
|
+
/** The platform's agent runs this judge, on the workspace's own serverless account. */
|
|
2600
|
+
auto: boolean;
|
|
2601
|
+
/**
|
|
2602
|
+
* Was automatic when the agent was turned off. Turning the agent back on
|
|
2603
|
+
* restores exactly these judges.
|
|
2604
|
+
*/
|
|
2605
|
+
auto_paused: boolean;
|
|
2606
|
+
/**
|
|
2607
|
+
* Why the last turn-on did NOT restore this judge. Null is the ordinary
|
|
2608
|
+
* state. Set when the resume declined to make it automatic because the model
|
|
2609
|
+
* it names is not one the serving gateway will route — handing it back would
|
|
2610
|
+
* buy a run that fails every conversation. It stays paused, so pointing it at
|
|
2611
|
+
* a model that routes and turning the agent on again brings it back.
|
|
2612
|
+
*/
|
|
2613
|
+
auto_pause_reason: string | null;
|
|
2614
|
+
created_by: string | null;
|
|
2615
|
+
created_at: string;
|
|
2616
|
+
updated_at: string;
|
|
2617
|
+
}
|
|
2618
|
+
export interface LoopJudgeParams {
|
|
2619
|
+
name: string;
|
|
2620
|
+
instructions: string;
|
|
2621
|
+
dimensions: LoopJudgeDimension[];
|
|
2622
|
+
selection?: LoopJudgeSelection;
|
|
2623
|
+
model?: string;
|
|
2624
|
+
/**
|
|
2625
|
+
* Let this judge write the answer that should have been given, not just a
|
|
2626
|
+
* score. Off by default: a reference answer written by a model and then
|
|
2627
|
+
* trained on is distillation, which is a decision you make deliberately.
|
|
2628
|
+
*/
|
|
2629
|
+
write_gold?: boolean;
|
|
2630
|
+
enabled?: boolean;
|
|
2631
|
+
/**
|
|
2632
|
+
* Hand the running of this judge to the platform's agent: new conversations
|
|
2633
|
+
* in its slice are scored as they arrive, one model call each, on THIS
|
|
2634
|
+
* WORKSPACE'S OWN serverless account (the agent spends through a managed key
|
|
2635
|
+
* of yours, "Conscious Loop" in your key list). Requires `model`. Off, you
|
|
2636
|
+
* run the model yourself with startRun / takeWork / postVerdicts.
|
|
2637
|
+
*/
|
|
2638
|
+
auto?: boolean;
|
|
2639
|
+
}
|
|
2640
|
+
/** One pass of one rubric over one slice. */
|
|
2641
|
+
export interface LoopJudgeRun {
|
|
2642
|
+
id: string;
|
|
2643
|
+
judge_id: string;
|
|
2644
|
+
judge_name?: string;
|
|
2645
|
+
/**
|
|
2646
|
+
* `abandoned` is a run you opened and never drained: nothing was scored on
|
|
2647
|
+
* it for a week and items were still waiting. Nothing is deleted and posting
|
|
2648
|
+
* verdicts to it still works and still closes it as done. It exists so that
|
|
2649
|
+
* `open` keeps meaning "somebody is working on this".
|
|
2650
|
+
*
|
|
2651
|
+
* `stopped` is a run somebody closed on purpose before it finished, with
|
|
2652
|
+
* `stopRun`. Separate from `done` because a run that covered three of forty
|
|
2653
|
+
* conversations did not finish its work: every score it recorded is kept
|
|
2654
|
+
* either way, and reading one as the other overstates what was evaluated.
|
|
2655
|
+
*/
|
|
2656
|
+
status: 'open' | 'done' | 'abandoned' | 'stopped';
|
|
2657
|
+
instructions: string;
|
|
2658
|
+
dimensions: LoopJudgeDimension[];
|
|
2659
|
+
model: string | null;
|
|
2660
|
+
selected: number;
|
|
2661
|
+
scored: number;
|
|
2662
|
+
failed: number;
|
|
2663
|
+
/** Who drains it: your own code ('caller') or the platform's agent ('platform'). */
|
|
2664
|
+
runner: 'caller' | 'platform';
|
|
2665
|
+
/** The agent's last complaint about this run, or null while it is working. */
|
|
2666
|
+
last_error: string | null;
|
|
2667
|
+
last_activity_at: string | null;
|
|
2668
|
+
created_by: string | null;
|
|
2669
|
+
created_at: string;
|
|
2670
|
+
finished_at: string | null;
|
|
2671
|
+
}
|
|
2672
|
+
/** The workspace's managed serverless key the agent spends through. Never the secret. */
|
|
2673
|
+
export interface LoopAgentCredential {
|
|
2674
|
+
workspace_id: string;
|
|
2675
|
+
key_id: string;
|
|
2676
|
+
key_prefix: string;
|
|
2677
|
+
created_by: string | null;
|
|
2678
|
+
created_at: string;
|
|
2679
|
+
updated_at: string;
|
|
2680
|
+
last_used_at: string | null;
|
|
2681
|
+
last_error: string | null;
|
|
2682
|
+
revoked_at: string | null;
|
|
2683
|
+
/**
|
|
2684
|
+
* The most the agent's key may spend on model calls in a calendar month,
|
|
2685
|
+
* in cents. `null` = no cap (the default): the agent can spend up to the
|
|
2686
|
+
* workspace's serverless balance. Once reached, the agent's model calls
|
|
2687
|
+
* are refused until next month and its runs pause with that reason.
|
|
2688
|
+
*/
|
|
2689
|
+
monthly_spend_cap_cents: number | null;
|
|
2690
|
+
}
|
|
2691
|
+
export interface LoopAgentStatus {
|
|
2692
|
+
/** An agent can exist in this environment at all. */
|
|
2693
|
+
available: boolean;
|
|
2694
|
+
/** Which piece is missing when it cannot. */
|
|
2695
|
+
reason: string;
|
|
2696
|
+
/** A worker has checked in within the last minute. */
|
|
2697
|
+
online: boolean;
|
|
2698
|
+
agent: {
|
|
2699
|
+
seen_at: string;
|
|
2700
|
+
passes: number;
|
|
2701
|
+
judge_items: number;
|
|
2702
|
+
samples: number;
|
|
2703
|
+
} | null;
|
|
2704
|
+
credential: LoopAgentCredential | null;
|
|
2705
|
+
open_judge_runs: number;
|
|
2706
|
+
open_sample_runs: number;
|
|
2707
|
+
auto_judges: number;
|
|
2708
|
+
}
|
|
2709
|
+
export interface LoopSampleSelection extends LoopJudgeSelection {
|
|
2710
|
+
/** Skip conversations that already have alternatives. */
|
|
2711
|
+
only_unsampled?: boolean;
|
|
2712
|
+
}
|
|
2713
|
+
export interface LoopSampleRunParams {
|
|
2714
|
+
/** The model that writes the alternatives. For distillation, the teacher. */
|
|
2715
|
+
model: string;
|
|
2716
|
+
/** Alternatives per conversation, 1-8. Default 4. */
|
|
2717
|
+
n?: number;
|
|
2718
|
+
/** 0-2. Default 0.8; 0 makes every sample the same answer. */
|
|
2719
|
+
temperature?: number;
|
|
2720
|
+
/** 16-8192. Default 1024. */
|
|
2721
|
+
max_tokens?: number;
|
|
2722
|
+
/** Score each sample with this judge as it is written. */
|
|
2723
|
+
judge_id?: string;
|
|
2724
|
+
selection?: LoopSampleSelection;
|
|
2725
|
+
}
|
|
2726
|
+
/** "Write N alternatives to each conversation in this slice, and score them." */
|
|
2727
|
+
export interface LoopSampleRun {
|
|
2728
|
+
id: string;
|
|
2729
|
+
workspace_id: string;
|
|
2730
|
+
model: string;
|
|
2731
|
+
n: number;
|
|
2732
|
+
temperature: number;
|
|
2733
|
+
max_tokens: number;
|
|
2734
|
+
judge_id: string | null;
|
|
2735
|
+
judge_name: string | null;
|
|
2736
|
+
selection: LoopSampleSelection;
|
|
2737
|
+
status: 'open' | 'done';
|
|
2738
|
+
selected: number;
|
|
2739
|
+
done: number;
|
|
2740
|
+
failed: number;
|
|
2741
|
+
samples: number;
|
|
2742
|
+
last_error: string | null;
|
|
2743
|
+
last_activity_at: string | null;
|
|
2744
|
+
created_by: string | null;
|
|
2745
|
+
created_at: string;
|
|
2746
|
+
finished_at: string | null;
|
|
2747
|
+
}
|
|
2748
|
+
/**
|
|
2749
|
+
* One conversation to score, already rendered into the prompt to send.
|
|
2750
|
+
*
|
|
2751
|
+
* Send `prompt` as-is. Assembling it yourself is how two callers end up giving
|
|
2752
|
+
* the same rubric different instructions, and two judges given different
|
|
2753
|
+
* instructions are not one judge.
|
|
2754
|
+
*/
|
|
2755
|
+
export interface LoopJudgeWorkItem {
|
|
2756
|
+
trace_id: string;
|
|
2757
|
+
messages: Array<{
|
|
2758
|
+
role: string;
|
|
2759
|
+
content?: string;
|
|
2760
|
+
}>;
|
|
2761
|
+
answer: string;
|
|
2762
|
+
prompt: string;
|
|
2763
|
+
}
|
|
2764
|
+
export interface LoopJudgeWork {
|
|
2765
|
+
run: LoopJudgeRun;
|
|
2766
|
+
items: LoopJudgeWorkItem[];
|
|
2767
|
+
remaining: number;
|
|
2768
|
+
}
|
|
2769
|
+
export type LoopJudgeRunItemStatus = 'pending' | 'scored' | 'failed';
|
|
2770
|
+
/** What happened to one conversation in one run. */
|
|
2771
|
+
export interface LoopJudgeRunItem {
|
|
2772
|
+
trace_id: string;
|
|
2773
|
+
status: LoopJudgeRunItemStatus;
|
|
2774
|
+
/**
|
|
2775
|
+
* Why this one could not be scored, as the caller reported it. This is where
|
|
2776
|
+
* a wrong model name or a refused key shows up; null for anything that is
|
|
2777
|
+
* not failed.
|
|
2778
|
+
*/
|
|
2779
|
+
error: string | null;
|
|
2780
|
+
scored_at: string | null;
|
|
2781
|
+
}
|
|
2782
|
+
export interface LoopJudgeRunItems {
|
|
2783
|
+
run: LoopJudgeRun;
|
|
2784
|
+
items: LoopJudgeRunItem[];
|
|
2785
|
+
/** True when the page ended before the run did. */
|
|
2786
|
+
has_more: boolean;
|
|
2787
|
+
/** Where to carry on from. Present only when `has_more`. */
|
|
2788
|
+
next_offset?: number;
|
|
2789
|
+
}
|
|
2790
|
+
/** One scored conversation going back. */
|
|
2791
|
+
export interface LoopJudgeVerdict {
|
|
2792
|
+
trace_id: string;
|
|
2793
|
+
/** One score per dimension the rubric asked for, each between 0 and 1. */
|
|
2794
|
+
scores?: Record<string, number>;
|
|
2795
|
+
/** Your own combination of them. Left out, the plain mean is used. */
|
|
2796
|
+
overall?: number;
|
|
2797
|
+
reason?: string;
|
|
2798
|
+
/** Only stored if the judge was created with `write_gold`. */
|
|
2799
|
+
gold?: string;
|
|
2800
|
+
/** Mark an item you could not score, instead of dropping it silently. */
|
|
2801
|
+
error?: string;
|
|
2802
|
+
}
|
|
2803
|
+
export interface LoopJudgeVerdictResult {
|
|
2804
|
+
run: LoopJudgeRun;
|
|
2805
|
+
recorded: number;
|
|
2806
|
+
failed: number;
|
|
2807
|
+
rejected: Array<{
|
|
2808
|
+
trace_id: string;
|
|
2809
|
+
error: string;
|
|
2810
|
+
}>;
|
|
2811
|
+
}
|
|
2812
|
+
export interface LoopStats {
|
|
2813
|
+
traces: number;
|
|
2814
|
+
signalled_traces: number;
|
|
2815
|
+
signals: number;
|
|
2816
|
+
by_verdict: Record<string, number>;
|
|
2817
|
+
by_source: Record<string, number>;
|
|
2818
|
+
datasets: number;
|
|
2819
|
+
/** Upper bounds: deduplication runs when a set is built. */
|
|
2820
|
+
ready: Record<LoopMethod, number>;
|
|
2821
|
+
}
|
|
2822
|
+
export interface LoopCandidateParams {
|
|
2823
|
+
completion?: string;
|
|
2824
|
+
/** An alternative that is itself a tool call. */
|
|
2825
|
+
tool_calls?: LoopToolCall[];
|
|
2826
|
+
/** Which model produced this alternative. The teacher, when distilling. */
|
|
2827
|
+
model?: string;
|
|
2828
|
+
model_version?: string;
|
|
2829
|
+
/** Unscored alternatives cannot pair -- a missing score is not a low one. */
|
|
2830
|
+
score?: number;
|
|
2831
|
+
/** Required alongside a score: human | verifier | judge | behavioural. */
|
|
2832
|
+
score_source?: LoopSource;
|
|
2833
|
+
reason?: string;
|
|
2834
|
+
metadata?: Record<string, unknown>;
|
|
2835
|
+
}
|
|
2836
|
+
export interface LoopCandidate {
|
|
2837
|
+
id: string;
|
|
2838
|
+
trace_id: string;
|
|
2839
|
+
model: string | null;
|
|
2840
|
+
model_version: string | null;
|
|
2841
|
+
completion: string;
|
|
2842
|
+
tool_calls?: LoopToolCall[];
|
|
2843
|
+
score: number | null;
|
|
2844
|
+
score_source: LoopSource | null;
|
|
2845
|
+
reason: string | null;
|
|
2846
|
+
metadata: Record<string, unknown>;
|
|
2847
|
+
created_at: string;
|
|
2848
|
+
}
|
|
2849
|
+
/**
|
|
2850
|
+
* The deterministic checks a grader can perform.
|
|
2851
|
+
*
|
|
2852
|
+
* `matches_gold` is the one kind with no expected value of its own: it
|
|
2853
|
+
* compares each answer to the gold answer recorded on that same conversation
|
|
2854
|
+
* (or the human correction when there is no gold), scores sampled
|
|
2855
|
+
* alternatives against the same gold, and is NOT APPLIED to a conversation
|
|
2856
|
+
* that carries neither, so it never marks down an unlabelled answer.
|
|
2857
|
+
*/
|
|
2858
|
+
export type LoopGraderKind = 'matches_gold' | 'exact_match' | 'contains' | 'regex' | 'json_valid' | 'numeric' | 'tool_called';
|
|
2859
|
+
export interface LoopGraderConfig {
|
|
2860
|
+
expected?: string;
|
|
2861
|
+
/** Every phrase that must appear. */
|
|
2862
|
+
required?: string[];
|
|
2863
|
+
/** Phrases that must not. On their own these are passed by silence. */
|
|
2864
|
+
forbidden?: string[];
|
|
2865
|
+
pattern?: string;
|
|
2866
|
+
/** Keys the answer must carry, for `json_valid`. */
|
|
2867
|
+
keys?: string[];
|
|
2868
|
+
/**
|
|
2869
|
+
* How far from `expected` still counts, for `numeric`. For `matches_gold`,
|
|
2870
|
+
* how far from the gold answer still counts when both are bare numbers.
|
|
2871
|
+
*/
|
|
2872
|
+
tolerance?: number;
|
|
2873
|
+
/** The function that must have been called, for `tool_called`. */
|
|
2874
|
+
function?: string;
|
|
2875
|
+
case_sensitive?: boolean;
|
|
2876
|
+
}
|
|
2877
|
+
export interface LoopGraderParams {
|
|
2878
|
+
name: string;
|
|
2879
|
+
kind: LoopGraderKind;
|
|
2880
|
+
config: LoopGraderConfig;
|
|
2881
|
+
/** The multiplier. Twice as important, twice the weight. Must exceed zero. */
|
|
2882
|
+
weight?: number;
|
|
2883
|
+
enabled?: boolean;
|
|
2884
|
+
/** Limit the rule to one capture source. Omit to apply it everywhere. */
|
|
2885
|
+
deployment_id?: string;
|
|
2886
|
+
/**
|
|
2887
|
+
* Aim the rule at one slice of the corpus. A master group covers its
|
|
2888
|
+
* children. Outside that slice the rule is NOT APPLIED, rather than failed,
|
|
2889
|
+
* so it stays out of the score entirely. Omit to apply it everywhere.
|
|
2890
|
+
*/
|
|
2891
|
+
label?: string;
|
|
2892
|
+
}
|
|
2893
|
+
export interface LoopGrader extends LoopGraderParams {
|
|
2894
|
+
id: string;
|
|
2895
|
+
workspace_id: string;
|
|
2896
|
+
weight: number;
|
|
2897
|
+
enabled: boolean;
|
|
2898
|
+
created_by: string | null;
|
|
2899
|
+
created_at: string;
|
|
2900
|
+
updated_at: string;
|
|
2901
|
+
}
|
|
2902
|
+
export interface LoopGraderResult {
|
|
2903
|
+
grader_id: string;
|
|
2904
|
+
name: string;
|
|
2905
|
+
kind: string;
|
|
2906
|
+
weight: number;
|
|
2907
|
+
passed: boolean;
|
|
2908
|
+
score: number;
|
|
2909
|
+
/** What happened, in words -- "failed" alone sends you to read the rule. */
|
|
2910
|
+
detail: string;
|
|
2911
|
+
/** The rule had no opinion. Excluded from the score rather than counted 0. */
|
|
2912
|
+
skipped: boolean;
|
|
2913
|
+
}
|
|
2914
|
+
export interface LoopGradeReport {
|
|
2915
|
+
/** Weighted mean over the rules that applied, 0 to 1. */
|
|
2916
|
+
score: number;
|
|
2917
|
+
results: LoopGraderResult[];
|
|
2918
|
+
/** The denominator. 1.0 from one rule is not 1.0 from six. */
|
|
2919
|
+
applied: number;
|
|
2920
|
+
skipped: number;
|
|
2921
|
+
}
|
|
2922
|
+
export interface LoopGradeResult {
|
|
2923
|
+
report: LoopGradeReport;
|
|
2924
|
+
/** Null when no rule applied, because no verdict was invented. */
|
|
2925
|
+
signal_id: string | null;
|
|
2926
|
+
candidates_graded: number;
|
|
2927
|
+
}
|
|
2928
|
+
export interface LoopLabel {
|
|
2929
|
+
label: string;
|
|
2930
|
+
/** The master group this label belongs to, if any. */
|
|
2931
|
+
parent: string | null;
|
|
2932
|
+
created_by: string | null;
|
|
2933
|
+
created_at: string;
|
|
2934
|
+
/** The dimension this value belongs to. `tag` for a bare name. */
|
|
2935
|
+
key: string;
|
|
2936
|
+
}
|
|
2937
|
+
/** One master group, and how many of a label's conversations are in it. */
|
|
2938
|
+
export interface LoopLabelParentCount {
|
|
2939
|
+
parent: string;
|
|
2940
|
+
traces: number;
|
|
2941
|
+
}
|
|
2942
|
+
export interface LoopLabelCount {
|
|
2943
|
+
label: string;
|
|
2944
|
+
/**
|
|
2945
|
+
* The master group ALL of this label's conversations are in, and nothing
|
|
2946
|
+
* else. Null when they disagree: the registry used to answer the
|
|
2947
|
+
* alphabetically last parent over a mixed group, so `billing` was reported
|
|
2948
|
+
* under `support` while seven of its twelve conversations were in no group
|
|
2949
|
+
* at all. Render "(in x)" from this field only.
|
|
2950
|
+
*/
|
|
2951
|
+
parent: string | null;
|
|
2952
|
+
/**
|
|
2953
|
+
* Every master group ANY of them are in, sorted, with how many of this
|
|
2954
|
+
* label's conversations are in each. Empty when there are none.
|
|
2955
|
+
*
|
|
2956
|
+
* Read this, not `parent`, to discover which master groups exist: a label
|
|
2957
|
+
* whose conversations disagree still belongs partly to a real group, and
|
|
2958
|
+
* `parent` is null for it, so a picker built from `parent` alone loses the
|
|
2959
|
+
* group along with the false claim. The count is per group rather than the
|
|
2960
|
+
* label's own total, because adding a label's whole count to its parent is
|
|
2961
|
+
* the same mistake one level down. What is in no group at all is `traces`
|
|
2962
|
+
* minus the sum of these.
|
|
2963
|
+
*/
|
|
2964
|
+
parents: LoopLabelParentCount[];
|
|
2965
|
+
traces: number;
|
|
2966
|
+
key: string;
|
|
2967
|
+
}
|
|
2968
|
+
export type TrainingCadence = 'none' | 'daily' | 'weekly';
|
|
2969
|
+
export type TrainingCombinator = 'and' | 'or';
|
|
2970
|
+
/** What answers the customer's traffic today, and therefore what promotion re-points. */
|
|
2971
|
+
export type ServingKind = 'serverless_slug' | 'deployment';
|
|
2972
|
+
export type TrainingRulePausedReason = 'consent_invalid' | 'insufficient_funds' | 'authorizer_not_member' | 'model_not_trainable' | 'platform_capacity'
|
|
2973
|
+
/** A peer was unreachable when the rule was saved, so it has not been validated yet. */
|
|
2974
|
+
| 'validation_pending';
|
|
2975
|
+
/**
|
|
2976
|
+
* How a training rule trains, on the wire of the automatic-training routes.
|
|
2977
|
+
*
|
|
2978
|
+
* DELIBERATELY NOT {@link TrainingMethod}. That union -- 'sft' | 'pt' -- is
|
|
2979
|
+
* training-service's, and the two services enumerate different things: a
|
|
2980
|
+
* training job can be a plain pre-training run, and a training rule cannot,
|
|
2981
|
+
* while a rule may ask for preference training and a job asks for that through
|
|
2982
|
+
* a different field. The values here are the loop service's own constants
|
|
2983
|
+
* (TrainingMethodSFT / TrainingMethodRLHF in
|
|
2984
|
+
* services/loop-service/cmd/training_types.go), which is what these routes
|
|
2985
|
+
* accept and return. Sharing the training-service union here advertised 'pt',
|
|
2986
|
+
* which this service refuses, and made 'rlhf', which it returns, unspellable.
|
|
2987
|
+
*
|
|
2988
|
+
* `rlhf_type` stays a plain string on purpose: the platform refuses it until
|
|
2989
|
+
* capabilities enable preference training, and an SDK union would have to be
|
|
2990
|
+
* republished to keep up with a server-side capability flag.
|
|
2991
|
+
*/
|
|
2992
|
+
export type LoopTrainingMethod = 'sft' | 'rlhf';
|
|
2993
|
+
export type TrainType = 'lora' | 'qlora' | 'full';
|
|
2994
|
+
export type GradersScope = 'source' | 'all' | 'none';
|
|
2995
|
+
export type TrainingTrigger = 'rows' | 'cadence' | 'both' | 'manual';
|
|
2996
|
+
export type TrainingRunState = 'built' | 'registering' | 'dataset_validating' | 'preflighting' | 'job_creating' | 'training' | 'checkpoint_ready' | 'candidate_booking' | 'candidate_running' | 'evaluating' | 'reported' | 'awaiting_review' | 'promoting' | 'waiting_funds' | 'promoted' | 'rejected' | 'expired' | 'failed' | 'budget_stopped' | 'cancelled' | 'superseded' | 'rolled_back';
|
|
2997
|
+
/** The closed set a run never leaves. */
|
|
2998
|
+
export declare const TERMINAL_RUN_STATES: readonly TrainingRunState[];
|
|
2999
|
+
export type TrainingVerdict = 'better' | 'not_better' | 'inconclusive' | 'not_evaluated';
|
|
3000
|
+
export type TrainingDecision = 'auto_promoted' | 'promoted' | 'rejected' | 'auto_rejected' | 'rolled_back';
|
|
3001
|
+
export type NotifyState = 'none' | 'pending' | 'sending' | 'sent' | 'suppressed' | 'failed';
|
|
3002
|
+
export type EvaluationStatus = 'open' | 'done' | 'failed' | 'budget_stopped';
|
|
3003
|
+
export type EvaluationItemStatus = 'pending' | 'working' | 'done' | 'failed';
|
|
3004
|
+
export type EvaluationWinner = 'candidate' | 'incumbent' | 'tie';
|
|
3005
|
+
/** Typed warning codes. Each surface renders its own plain sentence. */
|
|
3006
|
+
export type EvaluationWarningCode = 'holdout_too_small' | 'judge_unreliable' | 'graders_not_applied' | 'many_failures' | 'no_judge' | 'judge_labelled_training_rows' | 'capture_source_moved';
|
|
3007
|
+
export type ConsentVia = 'console' | 'api_key' | 'sdk' | 'mcp';
|
|
3008
|
+
/** A function an assistant turn asked to invoke. `arguments` is JSON-encoded. */
|
|
3009
|
+
export interface TrainingToolCall {
|
|
3010
|
+
id?: string;
|
|
3011
|
+
type?: string;
|
|
3012
|
+
function: {
|
|
3013
|
+
name: string;
|
|
3014
|
+
arguments: string;
|
|
3015
|
+
};
|
|
3016
|
+
}
|
|
3017
|
+
/** One conversational turn, in the chat-completions shape. */
|
|
3018
|
+
export interface TrainingMessage {
|
|
3019
|
+
role: string;
|
|
3020
|
+
content?: string;
|
|
3021
|
+
tool_calls?: TrainingToolCall[];
|
|
3022
|
+
tool_call_id?: string;
|
|
3023
|
+
name?: string;
|
|
3024
|
+
}
|
|
3025
|
+
/** What serves this model today. The name is also the alias a promotion writes. */
|
|
3026
|
+
export interface TrainingRuleServing {
|
|
3027
|
+
kind: ServingKind;
|
|
3028
|
+
name: string;
|
|
3029
|
+
/** Set only when kind is 'deployment'. */
|
|
3030
|
+
deployment_id: string | null;
|
|
3031
|
+
}
|
|
3032
|
+
/** One rung of a ranked GPU ladder. The provider is the neutral public brand. */
|
|
3033
|
+
export interface TrainingGPURung {
|
|
3034
|
+
gpu_type: string;
|
|
3035
|
+
gpu_count: number;
|
|
3036
|
+
provider: string;
|
|
3037
|
+
region: string;
|
|
3038
|
+
tier: string;
|
|
3039
|
+
}
|
|
3040
|
+
/** The consent object: one standing instruction to train, judge and possibly promote. */
|
|
3041
|
+
export interface TrainingRule {
|
|
3042
|
+
id: string;
|
|
3043
|
+
workspace_id: string;
|
|
3044
|
+
build_rule_id: string;
|
|
3045
|
+
name: string;
|
|
3046
|
+
enabled: boolean;
|
|
3047
|
+
/** Why the platform stopped firing this rule. Waiting must not look like broken. */
|
|
3048
|
+
paused_reason: TrainingRulePausedReason | null;
|
|
3049
|
+
cadence: TrainingCadence;
|
|
3050
|
+
cadence_hour_utc: number;
|
|
3051
|
+
cadence_weekday: number | null;
|
|
3052
|
+
combinator: TrainingCombinator;
|
|
3053
|
+
min_new_rows: number | null;
|
|
3054
|
+
/** Spreads firings across the hour so every daily rule does not land on one minute. */
|
|
3055
|
+
jitter_seconds: number;
|
|
3056
|
+
next_due_at: string | null;
|
|
3057
|
+
last_checked_at: string | null;
|
|
3058
|
+
last_fired_at: string | null;
|
|
3059
|
+
last_run_id: string | null;
|
|
3060
|
+
/** Rendered verbatim: "waiting: ...", "due, but ...", "fired: run 7 from ...". */
|
|
3061
|
+
last_reason: string | null;
|
|
3062
|
+
serving: TrainingRuleServing;
|
|
3063
|
+
model_id: string;
|
|
3064
|
+
/** A 40-hex commit, pinned when the rule is saved. */
|
|
3065
|
+
model_revision: string;
|
|
3066
|
+
training_method: LoopTrainingMethod;
|
|
3067
|
+
/** Refused by the platform until capabilities enable preference training. */
|
|
3068
|
+
rlhf_type: string | null;
|
|
3069
|
+
train_type: TrainType;
|
|
3070
|
+
config: Record<string, unknown>;
|
|
3071
|
+
train_gpu_priorities: TrainingGPURung[];
|
|
3072
|
+
train_max_price_hour_cents: number;
|
|
3073
|
+
deploy_gpu_priorities: TrainingGPURung[];
|
|
3074
|
+
deploy_max_price_hour_cents: number;
|
|
3075
|
+
training_ceiling_cents: number;
|
|
3076
|
+
candidate_ceiling_cents: number;
|
|
3077
|
+
/** A dollar figure, never a call count. */
|
|
3078
|
+
eval_ceiling_cents: number;
|
|
3079
|
+
monthly_ceiling_cents: number | null;
|
|
3080
|
+
eval_judge_id: string | null;
|
|
3081
|
+
eval_judge_model: string | null;
|
|
3082
|
+
eval_graders_scope: GradersScope;
|
|
3083
|
+
eval_max_rows: number;
|
|
3084
|
+
eval_max_tokens: number;
|
|
3085
|
+
min_holdout_rows: number;
|
|
3086
|
+
/**
|
|
3087
|
+
* The standing benchmark replayed on every run of this rule, beside the
|
|
3088
|
+
* per-run comparison and never instead of it. Null is the ordinary state.
|
|
3089
|
+
*
|
|
3090
|
+
* Set with `setTrainingRuleBenchmark`, not by an update to the rule: it is a
|
|
3091
|
+
* decision to replay a fixed set on every future run, and it has refusals of
|
|
3092
|
+
* its own. It raises no amount, so it does not invalidate consent.
|
|
3093
|
+
*/
|
|
3094
|
+
benchmark_id: string | null;
|
|
3095
|
+
auto_promote: boolean;
|
|
3096
|
+
promote_margin: number;
|
|
3097
|
+
promote_min_win_rate: number;
|
|
3098
|
+
/** A gate, not a badge. */
|
|
3099
|
+
min_judge_agreement: number;
|
|
3100
|
+
review_window_hours: number;
|
|
3101
|
+
candidate_boot_deadline_minutes: number;
|
|
3102
|
+
keep_candidate_warm_minutes: number;
|
|
3103
|
+
/** The member whose wallet pays, and the identity every peer call is stamped with. */
|
|
3104
|
+
authorized_by: string;
|
|
3105
|
+
terms_version: string;
|
|
3106
|
+
revision: number;
|
|
3107
|
+
/** Equal to revision only while the consent is current. */
|
|
3108
|
+
accepted_revision: number;
|
|
3109
|
+
accepted_at: string | null;
|
|
3110
|
+
deleted_at: string | null;
|
|
3111
|
+
created_at: string;
|
|
3112
|
+
updated_at: string;
|
|
3113
|
+
}
|
|
3114
|
+
/** One append-only record of a member agreeing to spend, with the words they read. */
|
|
3115
|
+
export interface TrainingRuleConsent {
|
|
3116
|
+
id: number;
|
|
3117
|
+
rule_id: string;
|
|
3118
|
+
workspace_id: string;
|
|
3119
|
+
revision: number;
|
|
3120
|
+
terms_version: string;
|
|
3121
|
+
terms_text: string;
|
|
3122
|
+
snapshot: Record<string, unknown>;
|
|
3123
|
+
accepted_by: string;
|
|
3124
|
+
accepted_via: ConsentVia;
|
|
3125
|
+
accepted_at: string;
|
|
3126
|
+
}
|
|
3127
|
+
export interface TrainingRulePreflightRefusal {
|
|
3128
|
+
stage: string;
|
|
3129
|
+
code: string;
|
|
3130
|
+
message: string;
|
|
3131
|
+
}
|
|
3132
|
+
/** One key that cannot follow a cutover, named so the caller can say which. */
|
|
3133
|
+
export interface TrainingRuleKeyRef {
|
|
3134
|
+
key_id: string;
|
|
3135
|
+
prefix: string;
|
|
3136
|
+
name: string;
|
|
3137
|
+
}
|
|
3138
|
+
export interface TrainingRuleKeyCheck {
|
|
3139
|
+
keys_missing_deployments_read: TrainingRuleKeyRef[];
|
|
3140
|
+
}
|
|
3141
|
+
/** The side-effect-free estimate, and the exact sentence the member will accept. */
|
|
3142
|
+
export interface TrainingRulePreflight {
|
|
3143
|
+
valid: boolean;
|
|
3144
|
+
model_revision: string;
|
|
3145
|
+
worst_hourly_training_cents: number;
|
|
3146
|
+
worst_hourly_candidate_cents: number;
|
|
3147
|
+
max_training_hours: number;
|
|
3148
|
+
max_candidate_hours: number;
|
|
3149
|
+
/** Informational. The fence on evaluation is the dollar ceiling, never this. */
|
|
3150
|
+
eval_calls_max: number;
|
|
3151
|
+
training_ceiling_cents: number;
|
|
3152
|
+
candidate_ceiling_cents: number;
|
|
3153
|
+
eval_ceiling_cents: number;
|
|
3154
|
+
warnings: string[];
|
|
3155
|
+
refusals: TrainingRulePreflightRefusal[];
|
|
3156
|
+
/**
|
|
3157
|
+
* Every peer the platform could not reach on this pass, one entry each, in
|
|
3158
|
+
* the same `{stage, code, message}` shape as a refusal.
|
|
3159
|
+
*
|
|
3160
|
+
* A warning, not a refusal: an unreachable peer judged nothing, so it never
|
|
3161
|
+
* refuses a save. It is still why the rule cannot fire — both
|
|
3162
|
+
* `worst_hourly_*` come back `0` — so `valid` is `false`. Key a "cannot
|
|
3163
|
+
* save" state on this when `refusals` is empty; the same sentences are also
|
|
3164
|
+
* in `warnings`, so render one or the other.
|
|
3165
|
+
*
|
|
3166
|
+
* Key it on THIS, not on `model_revision`. A degraded pass echoes back the
|
|
3167
|
+
* `model_revision` the request carried rather than emptying it, so
|
|
3168
|
+
* `if (!model_revision)` is false on exactly the passes it was meant to
|
|
3169
|
+
* catch. Use `unreachable.length > 0`, or `/^[0-9a-f]{40}$/`.
|
|
3170
|
+
*/
|
|
3171
|
+
unreachable: TrainingRulePreflightRefusal[];
|
|
3172
|
+
key_check: TrainingRuleKeyCheck;
|
|
3173
|
+
terms_text: string;
|
|
3174
|
+
terms_version: string;
|
|
3175
|
+
}
|
|
3176
|
+
export interface TrainingRuleBuildSpec {
|
|
3177
|
+
method: string;
|
|
3178
|
+
spec: Record<string, unknown>;
|
|
3179
|
+
}
|
|
3180
|
+
/**
|
|
3181
|
+
* The editable half of a build rule.
|
|
3182
|
+
*
|
|
3183
|
+
* Deliberately not `LoopBuildRuleParams`: that type requires `name` and
|
|
3184
|
+
* `method`, which the PUT route does not accept, so an edit written against it
|
|
3185
|
+
* would have to resend two fields the platform ignores and a caller could not
|
|
3186
|
+
* tell that changing them did nothing.
|
|
3187
|
+
*/
|
|
3188
|
+
export interface LoopBuildRuleUpdateParams {
|
|
3189
|
+
enabled?: boolean;
|
|
3190
|
+
/** Rows reviewed SINCE THE LAST BUILD before this fires again. Minimum 10. */
|
|
3191
|
+
min_new_rows?: number;
|
|
3192
|
+
/** The same selection `createDataset` takes, replayed verbatim. */
|
|
3193
|
+
spec?: LoopDatasetCreateParams;
|
|
3194
|
+
}
|
|
3195
|
+
/**
|
|
3196
|
+
* `?` means absent: leave that part of the sentence alone. `null` is only
|
|
3197
|
+
* allowed where the column is nullable and clearing it is a real edit, and it
|
|
3198
|
+
* is spelled out field by field rather than applied to the whole object.
|
|
3199
|
+
*/
|
|
3200
|
+
export interface TrainingRuleTriggerInput {
|
|
3201
|
+
cadence?: TrainingCadence;
|
|
3202
|
+
cadence_hour_utc?: number;
|
|
3203
|
+
/** null drops the weekday, which is what a weekly rule moved to daily needs. */
|
|
3204
|
+
cadence_weekday?: number | null;
|
|
3205
|
+
combinator?: TrainingCombinator;
|
|
3206
|
+
/** null removes the row floor, leaving the schedule as the only trigger. */
|
|
3207
|
+
min_new_rows?: number | null;
|
|
3208
|
+
}
|
|
3209
|
+
export interface TrainingRuleTrainingInput {
|
|
3210
|
+
model_id?: string;
|
|
3211
|
+
model_revision?: string;
|
|
3212
|
+
training_method?: LoopTrainingMethod;
|
|
3213
|
+
/**
|
|
3214
|
+
* A plain string, refused by the platform until capabilities enable
|
|
3215
|
+
* preference training. null clears it, which is what moving a rule back to
|
|
3216
|
+
* plain supervised training means.
|
|
3217
|
+
*/
|
|
3218
|
+
rlhf_type?: string | null;
|
|
3219
|
+
train_type?: TrainType;
|
|
3220
|
+
config?: Record<string, unknown>;
|
|
3221
|
+
train_gpu_priorities?: TrainingGPURung[];
|
|
3222
|
+
train_max_price_hour_cents?: number;
|
|
3223
|
+
}
|
|
3224
|
+
export interface TrainingRuleDeployInput {
|
|
3225
|
+
deploy_gpu_priorities?: TrainingGPURung[];
|
|
3226
|
+
deploy_max_price_hour_cents?: number;
|
|
3227
|
+
context_length?: number;
|
|
3228
|
+
quant?: string;
|
|
3229
|
+
serving_config?: Record<string, unknown>;
|
|
3230
|
+
hf_integration_id?: string;
|
|
3231
|
+
}
|
|
3232
|
+
export interface TrainingRuleMoneyInput {
|
|
3233
|
+
training_ceiling_cents?: number;
|
|
3234
|
+
candidate_ceiling_cents?: number;
|
|
3235
|
+
eval_ceiling_cents?: number;
|
|
3236
|
+
/** null removes the monthly cap. Absent leaves it exactly where it stands. */
|
|
3237
|
+
monthly_ceiling_cents?: number | null;
|
|
3238
|
+
eval_max_rows?: number;
|
|
3239
|
+
}
|
|
3240
|
+
export interface TrainingRuleEvaluationInput {
|
|
3241
|
+
/** null removes the judge from this rule. */
|
|
3242
|
+
judge_id?: string | null;
|
|
3243
|
+
/** null drops the override, returning to the judge's own advisory model. */
|
|
3244
|
+
judge_model?: string | null;
|
|
3245
|
+
graders_scope?: GradersScope;
|
|
3246
|
+
min_holdout_rows?: number;
|
|
3247
|
+
eval_max_tokens?: number;
|
|
3248
|
+
}
|
|
3249
|
+
export interface TrainingRulePromotionInput {
|
|
3250
|
+
auto_promote?: boolean;
|
|
3251
|
+
promote_margin?: number;
|
|
3252
|
+
promote_min_win_rate?: number;
|
|
3253
|
+
min_judge_agreement?: number;
|
|
3254
|
+
review_window_hours?: number;
|
|
3255
|
+
keep_candidate_warm_minutes?: number;
|
|
3256
|
+
}
|
|
3257
|
+
/**
|
|
3258
|
+
* The version of the terms the member read and accepted.
|
|
3259
|
+
*
|
|
3260
|
+
* Send back the `terms_version` the preflight returned. The SDK deliberately
|
|
3261
|
+
* ships no constant for it: a pinned version in a published package goes stale
|
|
3262
|
+
* the moment the platform revises the terms, and the value a member consented
|
|
3263
|
+
* to has to be the one they were actually shown.
|
|
3264
|
+
*/
|
|
3265
|
+
export interface TrainingRuleAcceptTerms {
|
|
3266
|
+
terms_version: string;
|
|
3267
|
+
}
|
|
3268
|
+
/** The create body without the yes. The preflight route takes exactly this. */
|
|
3269
|
+
export interface TrainingRulePreflightRequest {
|
|
3270
|
+
workspace_id?: string;
|
|
3271
|
+
name?: string;
|
|
3272
|
+
/** Exactly one of build_rule_id and build is given. */
|
|
3273
|
+
build_rule_id?: string;
|
|
3274
|
+
build?: TrainingRuleBuildSpec;
|
|
3275
|
+
trigger?: TrainingRuleTriggerInput;
|
|
3276
|
+
serving?: TrainingRuleServing;
|
|
3277
|
+
training?: TrainingRuleTrainingInput;
|
|
3278
|
+
deploy?: TrainingRuleDeployInput;
|
|
3279
|
+
money?: TrainingRuleMoneyInput;
|
|
3280
|
+
evaluation?: TrainingRuleEvaluationInput;
|
|
3281
|
+
promotion?: TrainingRulePromotionInput;
|
|
3282
|
+
enabled?: boolean;
|
|
3283
|
+
}
|
|
3284
|
+
export interface TrainingRuleCreateRequest extends TrainingRulePreflightRequest {
|
|
3285
|
+
/** Absent is a refusal, not a default. */
|
|
3286
|
+
accept_terms?: TrainingRuleAcceptTerms;
|
|
3287
|
+
/**
|
|
3288
|
+
* The standing benchmark this rule replays, set here so it applies to the
|
|
3289
|
+
* rule's FIRST run.
|
|
3290
|
+
*
|
|
3291
|
+
* A rule can fire within seconds of being created, and every step of a run
|
|
3292
|
+
* reads the benchmark off the rule as it stood at that moment. Attaching one
|
|
3293
|
+
* afterwards with `setTrainingRuleBenchmark` applies from the next run and
|
|
3294
|
+
* says nothing about the first, which is the run somebody is watching.
|
|
3295
|
+
*
|
|
3296
|
+
* Refused on the same terms the attach route refuses it: `404` when it is not
|
|
3297
|
+
* this workspace's, `409 BENCHMARK_RETIRED` when it has been retired. Use
|
|
3298
|
+
* `setTrainingRuleBenchmark` to change or remove it later; it is not on the
|
|
3299
|
+
* preflight body and not on the update body.
|
|
3300
|
+
*/
|
|
3301
|
+
benchmark_id?: string;
|
|
3302
|
+
}
|
|
3303
|
+
export interface TrainingRuleUpdateRequest extends TrainingRulePreflightRequest {
|
|
3304
|
+
expected_revision?: number;
|
|
3305
|
+
accept_terms?: TrainingRuleAcceptTerms;
|
|
3306
|
+
}
|
|
3307
|
+
export interface TrainingRuleConsentRequest {
|
|
3308
|
+
terms_version: string;
|
|
3309
|
+
revision: number;
|
|
3310
|
+
}
|
|
3311
|
+
export interface TrainingRuleListParams {
|
|
3312
|
+
enabled?: boolean;
|
|
3313
|
+
limit?: number;
|
|
3314
|
+
offset?: number;
|
|
3315
|
+
}
|
|
3316
|
+
export interface TrainingRunPromoteRequest {
|
|
3317
|
+
expected_revision?: number;
|
|
3318
|
+
/** Required to promote an inconclusive comparison. */
|
|
3319
|
+
force?: boolean;
|
|
3320
|
+
}
|
|
3321
|
+
export interface TrainingRunRejectRequest {
|
|
3322
|
+
reason?: string;
|
|
3323
|
+
}
|
|
3324
|
+
export interface TrainingRunRollbackRequest {
|
|
3325
|
+
reason?: string;
|
|
3326
|
+
}
|
|
3327
|
+
export interface TrainingRunCancelRequest {
|
|
3328
|
+
reason?: string;
|
|
3329
|
+
}
|
|
3330
|
+
export interface TrainingRunListParams {
|
|
3331
|
+
rule_id?: string;
|
|
3332
|
+
state?: TrainingRunState;
|
|
3333
|
+
limit?: number;
|
|
3334
|
+
offset?: number;
|
|
3335
|
+
}
|
|
3336
|
+
/** One firing, from the curated set to the decision. Money is frozen at fire time. */
|
|
3337
|
+
export interface TrainingRun {
|
|
3338
|
+
id: string;
|
|
3339
|
+
rule_id: string | null;
|
|
3340
|
+
rule_name: string | null;
|
|
3341
|
+
workspace_id: string;
|
|
3342
|
+
seq: number;
|
|
3343
|
+
rule_revision: number;
|
|
3344
|
+
authorized_by: string;
|
|
3345
|
+
rule_snapshot: Record<string, unknown>;
|
|
3346
|
+
trigger: TrainingTrigger;
|
|
3347
|
+
fired_reason: string | null;
|
|
3348
|
+
state: TrainingRunState;
|
|
3349
|
+
state_entered_at: string;
|
|
3350
|
+
attempts: number;
|
|
3351
|
+
next_poll_at: string | null;
|
|
3352
|
+
last_reason: string | null;
|
|
3353
|
+
error_code: string | null;
|
|
3354
|
+
last_error: string | null;
|
|
3355
|
+
training_ceiling_cents: number;
|
|
3356
|
+
candidate_ceiling_cents: number;
|
|
3357
|
+
eval_ceiling_cents: number;
|
|
3358
|
+
billed_training_cents: number;
|
|
3359
|
+
billed_candidate_cents: number;
|
|
3360
|
+
spent_eval_cents: number;
|
|
3361
|
+
loop_dataset_id: string | null;
|
|
3362
|
+
train_rows: number | null;
|
|
3363
|
+
holdout_rows: number | null;
|
|
3364
|
+
train_dataset_id: string | null;
|
|
3365
|
+
holdout_dataset_id: string | null;
|
|
3366
|
+
training_job_id: string | null;
|
|
3367
|
+
job_status: string | null;
|
|
3368
|
+
queue_deadline_at: string | null;
|
|
3369
|
+
checkpoint_id: string | null;
|
|
3370
|
+
checkpoint_step: number | null;
|
|
3371
|
+
training_eval_loss: number | null;
|
|
3372
|
+
candidate_deployment_id: string | null;
|
|
3373
|
+
candidate_name: string | null;
|
|
3374
|
+
candidate_seq: number;
|
|
3375
|
+
candidate_status: string | null;
|
|
3376
|
+
candidate_deadline_at: string | null;
|
|
3377
|
+
candidate_cleanup_at: string | null;
|
|
3378
|
+
evaluation_id: string | null;
|
|
3379
|
+
/** This run's replay of the rule's standing benchmark, if it had one. */
|
|
3380
|
+
benchmark_run_id: string | null;
|
|
3381
|
+
/**
|
|
3382
|
+
* What the replay's calls cost. Kept apart from `spent_eval_cents` so a
|
|
3383
|
+
* report can say what the comparison cost and what the benchmark cost; both
|
|
3384
|
+
* come out of `eval_ceiling_cents` and their sum can never exceed it, so
|
|
3385
|
+
* anything totalling what a run cost has to add this one too.
|
|
3386
|
+
*/
|
|
3387
|
+
benchmark_spent_cents: number;
|
|
3388
|
+
alias_name: string | null;
|
|
3389
|
+
alias_written_at: string | null;
|
|
3390
|
+
serving_before: TrainingRuleServing | null;
|
|
3391
|
+
rollback_available_until: string | null;
|
|
3392
|
+
verdict: TrainingVerdict | null;
|
|
3393
|
+
decision: TrainingDecision | null;
|
|
3394
|
+
decided_by: string | null;
|
|
3395
|
+
decided_at: string | null;
|
|
3396
|
+
auto_promote_at: string | null;
|
|
3397
|
+
review_deadline_at: string | null;
|
|
3398
|
+
notify_state: NotifyState;
|
|
3399
|
+
notify_event: string | null;
|
|
3400
|
+
notify_error: string | null;
|
|
3401
|
+
notify_attempts: number;
|
|
3402
|
+
created_at: string;
|
|
3403
|
+
updated_at: string;
|
|
3404
|
+
finished_at: string | null;
|
|
3405
|
+
}
|
|
3406
|
+
/** One row of the Runs table, carrying everything the list draws. */
|
|
3407
|
+
export interface TrainingRunSummary {
|
|
3408
|
+
id: string;
|
|
3409
|
+
rule_id: string | null;
|
|
3410
|
+
rule_name: string | null;
|
|
3411
|
+
workspace_id: string;
|
|
3412
|
+
seq: number;
|
|
3413
|
+
trigger: TrainingTrigger;
|
|
3414
|
+
state: TrainingRunState;
|
|
3415
|
+
state_entered_at: string;
|
|
3416
|
+
last_reason: string | null;
|
|
3417
|
+
error_code: string | null;
|
|
3418
|
+
verdict: TrainingVerdict | null;
|
|
3419
|
+
decision: TrainingDecision | null;
|
|
3420
|
+
train_rows: number | null;
|
|
3421
|
+
holdout_rows: number | null;
|
|
3422
|
+
billed_training_cents: number;
|
|
3423
|
+
billed_candidate_cents: number;
|
|
3424
|
+
spent_eval_cents: number;
|
|
3425
|
+
/** The fourth money column. A row that leaves it out adds up short. */
|
|
3426
|
+
benchmark_spent_cents: number;
|
|
3427
|
+
training_job_id: string | null;
|
|
3428
|
+
candidate_deployment_id: string | null;
|
|
3429
|
+
evaluation_id: string | null;
|
|
3430
|
+
review_deadline_at: string | null;
|
|
3431
|
+
auto_promote_at: string | null;
|
|
3432
|
+
notify_state: NotifyState;
|
|
3433
|
+
created_at: string;
|
|
3434
|
+
finished_at: string | null;
|
|
3435
|
+
}
|
|
3436
|
+
/** One line of the timeline. detail holds codes, ids and cents, never prompt text. */
|
|
3437
|
+
export interface TrainingRunEvent {
|
|
3438
|
+
seq: number;
|
|
3439
|
+
ts: string;
|
|
3440
|
+
from_state: TrainingRunState | null;
|
|
3441
|
+
to_state: TrainingRunState;
|
|
3442
|
+
reason: string;
|
|
3443
|
+
error_code: string | null;
|
|
3444
|
+
detail: Record<string, unknown>;
|
|
3445
|
+
actor: string;
|
|
3446
|
+
}
|
|
3447
|
+
export interface TrainingRunLinks {
|
|
3448
|
+
training_job_url: string | null;
|
|
3449
|
+
candidate_url: string | null;
|
|
3450
|
+
dataset_url: string | null;
|
|
3451
|
+
evaluation_url: string | null;
|
|
3452
|
+
/** The TREND the benchmark number belongs to, not the one replay. */
|
|
3453
|
+
benchmark_url: string | null;
|
|
3454
|
+
}
|
|
3455
|
+
/** What this reader may do right now. A button that cannot work is never shown. */
|
|
3456
|
+
export interface TrainingRunActions {
|
|
3457
|
+
promote: boolean;
|
|
3458
|
+
reject: boolean;
|
|
3459
|
+
rollback: boolean;
|
|
3460
|
+
cancel: boolean;
|
|
3461
|
+
}
|
|
3462
|
+
export interface EvaluationRef {
|
|
3463
|
+
kind: ServingKind;
|
|
3464
|
+
name: string;
|
|
3465
|
+
deployment_id: string | null;
|
|
3466
|
+
model_version: string | null;
|
|
3467
|
+
}
|
|
3468
|
+
/** Frozen on the evaluation. Both sides are regenerated with identical decoding. */
|
|
3469
|
+
export interface EvaluationDecoding {
|
|
3470
|
+
temperature: number;
|
|
3471
|
+
max_tokens: number;
|
|
3472
|
+
stream: boolean;
|
|
3473
|
+
}
|
|
3474
|
+
export interface EvaluationJudgeDimension {
|
|
3475
|
+
key: string;
|
|
3476
|
+
description: string;
|
|
3477
|
+
}
|
|
3478
|
+
export interface EvaluationDimensionScore {
|
|
3479
|
+
dimension: string;
|
|
3480
|
+
incumbent: number;
|
|
3481
|
+
candidate: number;
|
|
3482
|
+
delta: number;
|
|
3483
|
+
}
|
|
3484
|
+
export interface EvaluationGraderScore {
|
|
3485
|
+
grader_id: string;
|
|
3486
|
+
name: string;
|
|
3487
|
+
incumbent: number;
|
|
3488
|
+
candidate: number;
|
|
3489
|
+
delta: number;
|
|
3490
|
+
/** A grader that applied to four rows has not measured anything. */
|
|
3491
|
+
items_applied: number;
|
|
3492
|
+
}
|
|
3493
|
+
export interface EvaluationWarning {
|
|
3494
|
+
code: EvaluationWarningCode;
|
|
3495
|
+
message: string;
|
|
3496
|
+
}
|
|
3497
|
+
/** The thresholds this verdict was measured against, frozen with the report. */
|
|
3498
|
+
export interface EvaluationMargin {
|
|
3499
|
+
promote_margin: number;
|
|
3500
|
+
promote_min_win_rate: number;
|
|
3501
|
+
min_judge_agreement: number;
|
|
3502
|
+
min_holdout_rows: number;
|
|
3503
|
+
}
|
|
3504
|
+
/**
|
|
3505
|
+
* One rubric dimension of a judge-versus-human agreement.
|
|
3506
|
+
*
|
|
3507
|
+
* Deliberately not EvaluationDimensionScore: that type carries `incumbent` and
|
|
3508
|
+
* `candidate`, which are the two models being compared, and an agreement has
|
|
3509
|
+
* neither. `pairs` is per dimension because a reviewer who rated one dimension
|
|
3510
|
+
* and skipped another leaves a different denominator behind each number.
|
|
3511
|
+
*/
|
|
3512
|
+
export interface JudgeAgreementDimension {
|
|
3513
|
+
dimension: string;
|
|
3514
|
+
pairs: number;
|
|
3515
|
+
agreement: number | null;
|
|
3516
|
+
mean_abs_error: number | null;
|
|
3517
|
+
}
|
|
3518
|
+
/** How often this judge agreed with the workspace's own reviewers. */
|
|
3519
|
+
export interface JudgeAgreement {
|
|
3520
|
+
judge_id: string;
|
|
3521
|
+
pairs: number;
|
|
3522
|
+
agreement: number | null;
|
|
3523
|
+
mean_abs_error: number | null;
|
|
3524
|
+
per_dimension: JudgeAgreementDimension[];
|
|
3525
|
+
window_days: number;
|
|
3526
|
+
computed_at: string;
|
|
3527
|
+
/** "Not enough reviewer overlap yet" is an answer; 100% of two is not. */
|
|
3528
|
+
enough_pairs: boolean;
|
|
3529
|
+
}
|
|
3530
|
+
/** The comparison report: the candidate against what serves today, same rows. */
|
|
3531
|
+
export interface Evaluation {
|
|
3532
|
+
id: string;
|
|
3533
|
+
run_id: string;
|
|
3534
|
+
workspace_id: string;
|
|
3535
|
+
loop_dataset_id: string | null;
|
|
3536
|
+
split: string;
|
|
3537
|
+
incumbent_ref: EvaluationRef;
|
|
3538
|
+
candidate_ref: EvaluationRef;
|
|
3539
|
+
judge_id: string | null;
|
|
3540
|
+
judge_name: string | null;
|
|
3541
|
+
judge_model: string | null;
|
|
3542
|
+
judge_instructions: string | null;
|
|
3543
|
+
judge_dimensions: EvaluationJudgeDimension[];
|
|
3544
|
+
judge_system_prompt: string | null;
|
|
3545
|
+
grader_ids: string[];
|
|
3546
|
+
decoding: EvaluationDecoding;
|
|
3547
|
+
rows_selected: number;
|
|
3548
|
+
rows_scored: number;
|
|
3549
|
+
rows_failed: number;
|
|
3550
|
+
/** win_rate is wins / (wins + losses). Ties are excluded and reported separately. */
|
|
3551
|
+
wins: number | null;
|
|
3552
|
+
losses: number | null;
|
|
3553
|
+
ties: number | null;
|
|
3554
|
+
win_rate: number | null;
|
|
3555
|
+
incumbent_mean: number | null;
|
|
3556
|
+
candidate_mean: number | null;
|
|
3557
|
+
mean_delta: number | null;
|
|
3558
|
+
judge_incumbent_mean: number | null;
|
|
3559
|
+
judge_candidate_mean: number | null;
|
|
3560
|
+
grader_incumbent_mean: number | null;
|
|
3561
|
+
grader_candidate_mean: number | null;
|
|
3562
|
+
grader_items_applied: number;
|
|
3563
|
+
per_dimension: EvaluationDimensionScore[];
|
|
3564
|
+
per_grader: EvaluationGraderScore[];
|
|
3565
|
+
judge_agreement: JudgeAgreement | null;
|
|
3566
|
+
/** Informational: there is no incumbent counterpart to compare it against. */
|
|
3567
|
+
trainer_eval_loss: number | null;
|
|
3568
|
+
warnings: EvaluationWarning[];
|
|
3569
|
+
verdict: TrainingVerdict | null;
|
|
3570
|
+
verdict_reason: string | null;
|
|
3571
|
+
margin_used: EvaluationMargin | null;
|
|
3572
|
+
prompt_tokens: number;
|
|
3573
|
+
completion_tokens: number;
|
|
3574
|
+
spent_cents: number;
|
|
3575
|
+
status: EvaluationStatus;
|
|
3576
|
+
last_error: string | null;
|
|
3577
|
+
created_at: string;
|
|
3578
|
+
finished_at: string | null;
|
|
3579
|
+
}
|
|
3580
|
+
/** One model's answer to one held-out conversation, with the scores it earned. */
|
|
3581
|
+
export interface EvaluationItemSide {
|
|
3582
|
+
completion: string | null;
|
|
3583
|
+
tool_calls: TrainingToolCall[] | null;
|
|
3584
|
+
grader: Record<string, unknown> | null;
|
|
3585
|
+
judge: Record<string, unknown> | null;
|
|
3586
|
+
score: number | null;
|
|
3587
|
+
model_version: string | null;
|
|
3588
|
+
}
|
|
3589
|
+
/**
|
|
3590
|
+
* One paired conversation behind the numbers.
|
|
3591
|
+
*
|
|
3592
|
+
* trace_id is nullable on purpose: deleting one conversation must not shrink a
|
|
3593
|
+
* finished report so that its stated n and its visible rows disagree.
|
|
3594
|
+
*/
|
|
3595
|
+
export interface EvaluationItem {
|
|
3596
|
+
id: number;
|
|
3597
|
+
dataset_item_id: number;
|
|
3598
|
+
trace_id: string | null;
|
|
3599
|
+
trace_url: string | null;
|
|
3600
|
+
prompt: TrainingMessage[];
|
|
3601
|
+
tools: unknown[] | null;
|
|
3602
|
+
incumbent: EvaluationItemSide;
|
|
3603
|
+
candidate: EvaluationItemSide;
|
|
3604
|
+
human_verdict: number | null;
|
|
3605
|
+
winner: EvaluationWinner | null;
|
|
3606
|
+
status: EvaluationItemStatus;
|
|
3607
|
+
error: string | null;
|
|
3608
|
+
}
|
|
3609
|
+
export interface EvaluationItemListParams {
|
|
3610
|
+
winner?: EvaluationWinner;
|
|
3611
|
+
/** Capped at 100 by the service. */
|
|
3612
|
+
limit?: number;
|
|
3613
|
+
offset?: number;
|
|
3614
|
+
}
|
|
3615
|
+
export interface JudgeAgreementParams {
|
|
3616
|
+
/** RFC 3339. Narrows the window the agreement is computed over. */
|
|
3617
|
+
from?: string;
|
|
3618
|
+
to?: string;
|
|
3619
|
+
}
|
|
3620
|
+
/** Which model, whose words, how much. The cap is pushed to the workspace key. */
|
|
3621
|
+
export interface AgentSettings {
|
|
3622
|
+
workspace_id: string;
|
|
3623
|
+
default_model: string | null;
|
|
3624
|
+
judge_system_prompt: string | null;
|
|
3625
|
+
sampler_system_prompt: string | null;
|
|
3626
|
+
eval_monthly_cap_cents: number | null;
|
|
3627
|
+
updated_by: string | null;
|
|
3628
|
+
updated_at: string;
|
|
3629
|
+
}
|
|
3630
|
+
/** Absent leaves a setting alone; a present null returns it to the platform default. */
|
|
3631
|
+
export interface AgentSettingsRequest {
|
|
3632
|
+
default_model?: string | null;
|
|
3633
|
+
judge_system_prompt?: string | null;
|
|
3634
|
+
sampler_system_prompt?: string | null;
|
|
3635
|
+
eval_monthly_cap_cents?: number | null;
|
|
3636
|
+
}
|
|
3637
|
+
/** A re-pointable public handle. Promotion is one row write here. */
|
|
3638
|
+
export interface InferenceAlias {
|
|
3639
|
+
workspace_id: string;
|
|
3640
|
+
name: string;
|
|
3641
|
+
target_inference_id: string;
|
|
3642
|
+
previous_target_inference_id: string | null;
|
|
3643
|
+
/** Set only at adoption, when the alias takes over a deployment's own name. */
|
|
3644
|
+
shadows_inference_id: string | null;
|
|
3645
|
+
set_by: string;
|
|
3646
|
+
origin: string | null;
|
|
3647
|
+
created_at: string;
|
|
3648
|
+
updated_at: string;
|
|
3649
|
+
}
|
|
3650
|
+
export interface InferenceAliasRequest {
|
|
3651
|
+
target_inference_id: string;
|
|
3652
|
+
origin?: string;
|
|
3653
|
+
}
|
|
3654
|
+
/**
|
|
3655
|
+
* A benchmark is active until it is retired. There is no delete and no update:
|
|
3656
|
+
* the series of numbers measured against a set is what a benchmark is for, so
|
|
3657
|
+
* an edit would make every number before it incomparable with every number
|
|
3658
|
+
* after it, and a delete throws the series away.
|
|
3659
|
+
*/
|
|
3660
|
+
export type BenchmarkStatus = 'active' | 'retired';
|
|
3661
|
+
/** Where the pinned conversations are copied from. Read once, at creation. */
|
|
3662
|
+
export type BenchmarkSourceKind = 'traces' | 'dataset';
|
|
3663
|
+
/**
|
|
3664
|
+
* The life of one replay.
|
|
3665
|
+
*
|
|
3666
|
+
* `budget_stopped` reached the amount left for it inside the run's own ceiling,
|
|
3667
|
+
* and `abandoned` ran out of the time the run sets aside for it. Both are
|
|
3668
|
+
* points on the trend carrying `status_reason`, never silent gaps.
|
|
3669
|
+
*/
|
|
3670
|
+
export type BenchmarkRunStatus = 'open' | 'done' | 'failed' | 'budget_stopped' | 'abandoned';
|
|
3671
|
+
/**
|
|
3672
|
+
* How the pinned conversations become one number, frozen on the benchmark.
|
|
3673
|
+
*
|
|
3674
|
+
* `require_all_rows` is what makes the number comparable at all: a mean over
|
|
3675
|
+
* whichever conversations happened to succeed is a measurement of a different
|
|
3676
|
+
* set, so a short replay publishes no score and says why. A partial benchmark
|
|
3677
|
+
* is a missing number, never a lower one.
|
|
3678
|
+
*/
|
|
3679
|
+
export interface BenchmarkScoring {
|
|
3680
|
+
metric: string;
|
|
3681
|
+
scale: string;
|
|
3682
|
+
aggregate: string;
|
|
3683
|
+
require_all_rows: boolean;
|
|
3684
|
+
}
|
|
3685
|
+
/**
|
|
3686
|
+
* One rubric dimension of one replay, for both models.
|
|
3687
|
+
*
|
|
3688
|
+
* Deliberately not `EvaluationDimensionScore`: these are absolute means on a
|
|
3689
|
+
* fixed set and those are paired means on that run's own held-back rows.
|
|
3690
|
+
*/
|
|
3691
|
+
export interface BenchmarkDimensionScore {
|
|
3692
|
+
dimension: string;
|
|
3693
|
+
incumbent: number;
|
|
3694
|
+
candidate: number;
|
|
3695
|
+
delta: number;
|
|
3696
|
+
}
|
|
3697
|
+
/**
|
|
3698
|
+
* The frozen definition: the oracle, the scoring rule and the fingerprint of
|
|
3699
|
+
* the set. The conversations themselves are their own paged route.
|
|
3700
|
+
*/
|
|
3701
|
+
export interface Benchmark {
|
|
3702
|
+
id: string;
|
|
3703
|
+
workspace_id: string;
|
|
3704
|
+
name: string;
|
|
3705
|
+
status: BenchmarkStatus;
|
|
3706
|
+
/** Provenance only: everything needed from the judge is copied below it. */
|
|
3707
|
+
judge_id: string | null;
|
|
3708
|
+
judge_name: string;
|
|
3709
|
+
judge_model: string;
|
|
3710
|
+
judge_instructions: string;
|
|
3711
|
+
judge_dimensions: EvaluationJudgeDimension[];
|
|
3712
|
+
judge_system_prompt: string | null;
|
|
3713
|
+
decoding: EvaluationDecoding;
|
|
3714
|
+
scoring: BenchmarkScoring;
|
|
3715
|
+
item_count: number;
|
|
3716
|
+
/** Recomputed from the rows and compared at the start of every replay. */
|
|
3717
|
+
items_digest: string;
|
|
3718
|
+
/** A cap INSIDE the run's own amount for judge calls, never an addition to it. */
|
|
3719
|
+
per_run_ceiling_cents: number;
|
|
3720
|
+
created_by: string;
|
|
3721
|
+
created_at: string;
|
|
3722
|
+
retired_at: string | null;
|
|
3723
|
+
retired_by: string | null;
|
|
3724
|
+
}
|
|
3725
|
+
/**
|
|
3726
|
+
* One pinned conversation.
|
|
3727
|
+
*
|
|
3728
|
+
* `source_trace_id` is a link and is allowed to go null: the prompt was copied
|
|
3729
|
+
* at pin time, so retention removing the conversation takes away the ability to
|
|
3730
|
+
* open it and takes away nothing else. Every number already measured stays
|
|
3731
|
+
* exactly as comparable as it was.
|
|
3732
|
+
*/
|
|
3733
|
+
export interface BenchmarkItem {
|
|
3734
|
+
id: number;
|
|
3735
|
+
ordinal: number;
|
|
3736
|
+
source_trace_id: string | null;
|
|
3737
|
+
source_trace_url: string | null;
|
|
3738
|
+
prompt: TrainingMessage[];
|
|
3739
|
+
tools: unknown[] | null;
|
|
3740
|
+
}
|
|
3741
|
+
/**
|
|
3742
|
+
* One replay: this benchmark, on this training run, against both models.
|
|
3743
|
+
*
|
|
3744
|
+
* IT DOES NOT GATE PROMOTION. The paired comparison applies the judge to both
|
|
3745
|
+
* sides of one conversation in one pass, so most of the judge's variance
|
|
3746
|
+
* cancels in its delta; an absolute mean carries that variance whole. What
|
|
3747
|
+
* ships here is the number, the difference against what serves today on the
|
|
3748
|
+
* same set, and the history. A person reads the trend.
|
|
3749
|
+
*/
|
|
3750
|
+
export interface BenchmarkRun {
|
|
3751
|
+
id: string;
|
|
3752
|
+
benchmark_id: string;
|
|
3753
|
+
benchmark_name: string;
|
|
3754
|
+
run_id: string;
|
|
3755
|
+
workspace_id: string;
|
|
3756
|
+
incumbent_ref: EvaluationRef;
|
|
3757
|
+
candidate_ref: EvaluationRef;
|
|
3758
|
+
/** The set actually scored, recorded beside the score. */
|
|
3759
|
+
items_digest: string;
|
|
3760
|
+
rows_total: number;
|
|
3761
|
+
rows_scored: number;
|
|
3762
|
+
rows_failed: number;
|
|
3763
|
+
/** Absolute, on the frozen judge's scale, and null unless every row scored. */
|
|
3764
|
+
candidate_score: number | null;
|
|
3765
|
+
incumbent_score: number | null;
|
|
3766
|
+
score_delta: number | null;
|
|
3767
|
+
per_dimension: BenchmarkDimensionScore[];
|
|
3768
|
+
prompt_tokens: number;
|
|
3769
|
+
completion_tokens: number;
|
|
3770
|
+
spent_cents: number;
|
|
3771
|
+
ceiling_cents: number;
|
|
3772
|
+
status: BenchmarkRunStatus;
|
|
3773
|
+
/** The whole explanation when there is no score, so never empty on one. */
|
|
3774
|
+
status_reason: string | null;
|
|
3775
|
+
created_at: string;
|
|
3776
|
+
finished_at: string | null;
|
|
3777
|
+
}
|
|
3778
|
+
/**
|
|
3779
|
+
* One point on the trend line.
|
|
3780
|
+
*
|
|
3781
|
+
* It carries the checkpoint and what the person then decided, because a trend
|
|
3782
|
+
* with no idea what changed between two points is a chart rather than an
|
|
3783
|
+
* answer.
|
|
3784
|
+
*/
|
|
3785
|
+
export interface BenchmarkHistoryPoint {
|
|
3786
|
+
benchmark_run_id: string;
|
|
3787
|
+
run_id: string;
|
|
3788
|
+
run_seq: number;
|
|
3789
|
+
rule_id: string | null;
|
|
3790
|
+
rule_name: string | null;
|
|
3791
|
+
created_at: string;
|
|
3792
|
+
candidate_score: number | null;
|
|
3793
|
+
incumbent_score: number | null;
|
|
3794
|
+
score_delta: number | null;
|
|
3795
|
+
checkpoint_id: string | null;
|
|
3796
|
+
decision: TrainingDecision | null;
|
|
3797
|
+
status: BenchmarkRunStatus;
|
|
3798
|
+
status_reason: string | null;
|
|
3799
|
+
}
|
|
3800
|
+
/** Where the conversations are copied FROM. Read once, at creation, never again. */
|
|
3801
|
+
export interface BenchmarkSource {
|
|
3802
|
+
kind: BenchmarkSourceKind;
|
|
3803
|
+
trace_ids?: string[];
|
|
3804
|
+
dataset_id?: string;
|
|
3805
|
+
split?: string;
|
|
3806
|
+
}
|
|
3807
|
+
/**
|
|
3808
|
+
* The pin: 10 to 200 conversations, and a judge with a rubric to measure them.
|
|
3809
|
+
*
|
|
3810
|
+
* A conversation with nothing to ask a model is refused rather than skipped.
|
|
3811
|
+
* Pinning 47 of the 50 somebody chose is the set being wrong from the first
|
|
3812
|
+
* day, and they would never find out.
|
|
3813
|
+
*/
|
|
3814
|
+
export interface BenchmarkCreateParams {
|
|
3815
|
+
name: string;
|
|
3816
|
+
judge_id: string;
|
|
3817
|
+
source: BenchmarkSource;
|
|
3818
|
+
/** Absent uses the judge's own model, and absent that the workspace default. */
|
|
3819
|
+
judge_model?: string;
|
|
3820
|
+
/** What both models are given to answer in. A shorter answer is a different answer. */
|
|
3821
|
+
max_tokens?: number;
|
|
3822
|
+
per_run_ceiling_cents: number;
|
|
3823
|
+
}
|
|
3824
|
+
export interface BenchmarkListParams {
|
|
3825
|
+
status?: BenchmarkStatus;
|
|
3826
|
+
limit?: number;
|
|
3827
|
+
offset?: number;
|
|
3828
|
+
}
|
|
3829
|
+
export interface BenchmarkItemListParams {
|
|
3830
|
+
/** Capped at 100 by the service. */
|
|
3831
|
+
limit?: number;
|
|
3832
|
+
offset?: number;
|
|
3833
|
+
}
|
|
3834
|
+
export interface BenchmarkHistoryParams {
|
|
3835
|
+
limit?: number;
|
|
3836
|
+
offset?: number;
|
|
3837
|
+
}
|
|
3838
|
+
/** A reason is a courtesy here, not a requirement. */
|
|
3839
|
+
export interface BenchmarkRetireParams {
|
|
3840
|
+
reason?: string;
|
|
3841
|
+
}
|
|
3842
|
+
/**
|
|
3843
|
+
* Attach a benchmark to a rule, or detach it with a present null.
|
|
3844
|
+
*
|
|
3845
|
+
* Not optional, and not omittable: this route sets the field, so an absent key
|
|
3846
|
+
* would be a request with nothing in it. Send the id to attach, null to detach.
|
|
3847
|
+
*/
|
|
3848
|
+
export interface TrainingRuleBenchmarkRequest {
|
|
3849
|
+
benchmark_id: string | null;
|
|
3850
|
+
}
|
|
3851
|
+
export interface TrainingRuleListResponse {
|
|
3852
|
+
rules: TrainingRule[];
|
|
3853
|
+
total: number;
|
|
3854
|
+
}
|
|
3855
|
+
export interface TrainingRuleResponse {
|
|
3856
|
+
rule: TrainingRule;
|
|
3857
|
+
recent_runs: TrainingRunSummary[];
|
|
3858
|
+
month_spent_cents: number;
|
|
3859
|
+
judge_agreement: JudgeAgreement | null;
|
|
3860
|
+
}
|
|
3861
|
+
export interface TrainingRuleMutationResponse {
|
|
3862
|
+
rule: TrainingRule;
|
|
3863
|
+
/** The rule will not fire again until someone confirms the new amounts. */
|
|
3864
|
+
consent_required: boolean;
|
|
3865
|
+
preflight: TrainingRulePreflight | null;
|
|
3866
|
+
}
|
|
3867
|
+
export interface TrainingRuleDeleteResponse {
|
|
3868
|
+
deleted: boolean;
|
|
3869
|
+
rule_id: string;
|
|
3870
|
+
/**
|
|
3871
|
+
* The name this delete just spent. The delete is a soft delete and the name
|
|
3872
|
+
* index carries no partial predicate, so the row goes on holding the name
|
|
3873
|
+
* after it has left every list you can read, and no later rule in the
|
|
3874
|
+
* workspace can be called that. There is no purge and no undelete.
|
|
3875
|
+
*/
|
|
3876
|
+
retained_name: string;
|
|
3877
|
+
cancelled_run_id: string | null;
|
|
3878
|
+
}
|
|
3879
|
+
export interface TrainingRunListResponse {
|
|
3880
|
+
runs: TrainingRunSummary[];
|
|
3881
|
+
total: number;
|
|
3882
|
+
}
|
|
3883
|
+
export interface TrainingRunResponse {
|
|
3884
|
+
run: TrainingRun;
|
|
3885
|
+
timeline: TrainingRunEvent[];
|
|
3886
|
+
/**
|
|
3887
|
+
* The standing benchmark's two absolute numbers for this run, beside the
|
|
3888
|
+
* paired verdict and never in place of it. Null when no benchmark is
|
|
3889
|
+
* attached to the rule.
|
|
3890
|
+
*/
|
|
3891
|
+
benchmark: BenchmarkRun | null;
|
|
3892
|
+
links: TrainingRunLinks;
|
|
3893
|
+
available_actions: TrainingRunActions;
|
|
3894
|
+
}
|
|
3895
|
+
export interface TrainingRunActionResponse {
|
|
3896
|
+
run: TrainingRun;
|
|
3897
|
+
/** Set on rollback, the one action that changes what answers the traffic. */
|
|
3898
|
+
serving: TrainingRuleServing | null;
|
|
3899
|
+
}
|
|
3900
|
+
export interface EvaluationItemsResponse {
|
|
3901
|
+
items: EvaluationItem[];
|
|
3902
|
+
total: number;
|
|
3903
|
+
}
|
|
3904
|
+
export interface AgentSettingsResponse {
|
|
3905
|
+
settings: AgentSettings;
|
|
3906
|
+
}
|
|
3907
|
+
export interface InferenceAliasListResponse {
|
|
3908
|
+
aliases: InferenceAlias[];
|
|
3909
|
+
}
|
|
3910
|
+
export interface InferenceAliasResponse {
|
|
3911
|
+
alias: InferenceAlias;
|
|
3912
|
+
}
|
|
3913
|
+
export interface InferenceAliasDeleteResponse {
|
|
3914
|
+
deleted: boolean;
|
|
3915
|
+
}
|
|
3916
|
+
export interface BenchmarkListResponse {
|
|
3917
|
+
benchmarks: Benchmark[];
|
|
3918
|
+
total: number;
|
|
3919
|
+
}
|
|
3920
|
+
export interface BenchmarkItemsResponse {
|
|
3921
|
+
items: BenchmarkItem[];
|
|
3922
|
+
total: number;
|
|
3923
|
+
}
|
|
3924
|
+
/** The trend, newest first. Every replay is a point, including the scoreless ones. */
|
|
3925
|
+
export interface BenchmarkHistoryResponse {
|
|
3926
|
+
benchmark_id: string;
|
|
3927
|
+
points: BenchmarkHistoryPoint[];
|
|
3928
|
+
total: number;
|
|
3929
|
+
}
|