runbios-sdk 0.2.1-dev.97 → 0.2.1-rc.118
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +32 -2
- package/dist/client.js +1 -1
- package/dist/index.d.ts +12 -2
- package/dist/index.js +13 -1
- package/dist/resources/datasets.d.ts +8 -4
- package/dist/resources/datasets.js +15 -4
- package/dist/resources/inference.d.ts +7 -1
- package/dist/resources/inference.js +1 -1
- package/dist/resources/integrations.d.ts +21 -0
- package/dist/resources/integrations.js +20 -0
- package/dist/resources/loop.d.ts +378 -0
- package/dist/resources/loop.js +530 -0
- package/dist/resources/training.d.ts +20 -8
- package/dist/resources/training.js +48 -9
- package/dist/types.d.ts +620 -6
- package/package.json +2 -2
package/dist/types.d.ts
CHANGED
|
@@ -8,13 +8,13 @@ export interface BiOSConfig {
|
|
|
8
8
|
orgId?: string;
|
|
9
9
|
/** Workspace ID. Optional when using API keys (resolved from the key). Can override for multi-workspace keys. */
|
|
10
10
|
workspaceId?: string;
|
|
11
|
-
/** Base URL for the API. Falls back to the RUNBIOS_BASE_URL environment variable (legacy: BIOS_BASE_URL), then the canonical https://api-
|
|
11
|
+
/** Base URL for the API. Falls back to the RUNBIOS_BASE_URL environment variable (legacy: BIOS_BASE_URL), then the canonical https://api-staging.runbios.ai hostname. */
|
|
12
12
|
baseUrl?: string;
|
|
13
13
|
/** Request timeout in milliseconds. Defaults to 30000. */
|
|
14
14
|
timeout?: number;
|
|
15
15
|
/** Default per-deployment inference key. Can be overridden per inference call. */
|
|
16
16
|
inferenceKey?: string;
|
|
17
|
-
/** Inference base URL. Defaults to baseUrl, then https://api-
|
|
17
|
+
/** Inference base URL. Defaults to baseUrl, then https://api-staging.runbios.ai. */
|
|
18
18
|
inferenceBaseUrl?: string;
|
|
19
19
|
/** End-to-end inference timeout in milliseconds. Defaults to 15 minutes. */
|
|
20
20
|
inferenceTimeout?: number;
|
|
@@ -396,8 +396,9 @@ export interface DatasetPreview {
|
|
|
396
396
|
}
|
|
397
397
|
/** Parameters for importing a dataset from HuggingFace Hub. */
|
|
398
398
|
export interface DatasetImportHFParams {
|
|
399
|
-
/** HuggingFace dataset repository ID (e.g. "
|
|
399
|
+
/** HuggingFace dataset repository ID (e.g. "HuggingFaceH4/ultrachat_200k"). */
|
|
400
400
|
repoId: string;
|
|
401
|
+
revision?: string;
|
|
401
402
|
/** Display name for the imported dataset. */
|
|
402
403
|
name?: string;
|
|
403
404
|
/** Dataset subset/config to import. */
|
|
@@ -418,6 +419,7 @@ export interface DatasetImportHFParams {
|
|
|
418
419
|
/** Parameters for registering a HuggingFace dataset without an integration. */
|
|
419
420
|
export interface DatasetRegisterHFParams {
|
|
420
421
|
repoId: string;
|
|
422
|
+
revision?: string;
|
|
421
423
|
name?: string;
|
|
422
424
|
workspaceId?: string;
|
|
423
425
|
description?: string;
|
|
@@ -441,6 +443,7 @@ export interface DatasetHubSearchParams {
|
|
|
441
443
|
export interface DatasetHubPreviewParams {
|
|
442
444
|
/** HuggingFace dataset ID. */
|
|
443
445
|
datasetId: string;
|
|
446
|
+
integrationId?: string;
|
|
444
447
|
/** Split to preview. Defaults to "train". */
|
|
445
448
|
split?: string;
|
|
446
449
|
/** Subset/config name. */
|
|
@@ -565,8 +568,10 @@ export interface TrainingCreateParams {
|
|
|
565
568
|
name?: string;
|
|
566
569
|
/** Target workspace ID. */
|
|
567
570
|
workspaceId?: string;
|
|
568
|
-
/** Number of training epochs. */
|
|
571
|
+
/** Number of training epochs. Governs the full run when maxSteps is omitted. */
|
|
569
572
|
epochs?: number;
|
|
573
|
+
/** Optional hard step cap that overrides epochs. Omit for no cap and all configured epochs. */
|
|
574
|
+
maxSteps?: number;
|
|
570
575
|
/** Training batch size per device. */
|
|
571
576
|
batchSize?: number;
|
|
572
577
|
/** Gradient accumulation steps. */
|
|
@@ -609,7 +614,7 @@ export interface TrainingCreateParams {
|
|
|
609
614
|
integrationId?: string;
|
|
610
615
|
/** Existing network volume to attach. */
|
|
611
616
|
networkVolumeId?: string;
|
|
612
|
-
/**
|
|
617
|
+
/** @deprecated No effect. Prepared data is transient; exact resume rebuilds it from pinned source metadata and verifies its checksum. */
|
|
613
618
|
cacheDataset?: boolean;
|
|
614
619
|
/** Legacy dataset ordering mode. Prefer mixing for weighted/phased plans. */
|
|
615
620
|
datasetMixing?: 'shuffle' | 'sequential' | 'interleave' | 'random' | 'curriculum';
|
|
@@ -754,12 +759,37 @@ export interface TrainingListResponse {
|
|
|
754
759
|
limit: number;
|
|
755
760
|
offset: number;
|
|
756
761
|
}
|
|
762
|
+
export interface TrainingDeviceMetrics {
|
|
763
|
+
memory_used_mib: number | null;
|
|
764
|
+
memory_total_mib: number | null;
|
|
765
|
+
utilization_pct: number | null;
|
|
766
|
+
temperature_c: number | null;
|
|
767
|
+
}
|
|
768
|
+
export interface TrainingResourceMetrics {
|
|
769
|
+
scope: 'worker';
|
|
770
|
+
sampled_at: string | null;
|
|
771
|
+
received_at: string;
|
|
772
|
+
gpu_count: number | null;
|
|
773
|
+
gpu_memory_used_mib: number | null;
|
|
774
|
+
gpu_memory_total_mib: number | null;
|
|
775
|
+
gpu_utilization_pct: number | null;
|
|
776
|
+
gpu_temperature_c: number | null;
|
|
777
|
+
host_ram_used_gib: number | null;
|
|
778
|
+
host_ram_limit_gib: number | null;
|
|
779
|
+
host_ram_pct: number | null;
|
|
780
|
+
host_cpu_usage_pct: number | null;
|
|
781
|
+
host_cpu_limit_cores: number | null;
|
|
782
|
+
disk_used_gib: number | null;
|
|
783
|
+
disk_total_gib: number | null;
|
|
784
|
+
gpus: TrainingDeviceMetrics[];
|
|
785
|
+
}
|
|
757
786
|
/** Training metrics for a job. */
|
|
758
787
|
export interface TrainingMetrics {
|
|
759
788
|
metrics: MetricPoint[];
|
|
760
789
|
training_method?: string | null;
|
|
761
790
|
rlhf_type?: string | null;
|
|
762
791
|
graph_configs: MetricGraphConfig[];
|
|
792
|
+
resource_metrics?: TrainingResourceMetrics | null;
|
|
763
793
|
/** @deprecated Use metrics. Populated as a compatibility alias. */
|
|
764
794
|
steps?: MetricPoint[];
|
|
765
795
|
}
|
|
@@ -877,6 +907,8 @@ export interface TrainingPreflightDataset {
|
|
|
877
907
|
export interface TrainingPreflightWarning {
|
|
878
908
|
code: string;
|
|
879
909
|
message: string;
|
|
910
|
+
/** Request field the warning is about (e.g. `per_device_train_batch_size`), when there is one. */
|
|
911
|
+
field?: string;
|
|
880
912
|
}
|
|
881
913
|
/** Side-effect-free validation/sizing result; this endpoint never creates or bills a job. */
|
|
882
914
|
export interface TrainingPreflightResponse {
|
|
@@ -897,6 +929,131 @@ export interface TrainingPreflightResponse {
|
|
|
897
929
|
queue_eligible: boolean;
|
|
898
930
|
warnings: TrainingPreflightWarning[];
|
|
899
931
|
checked_at: string;
|
|
932
|
+
/**
|
|
933
|
+
* The trainer image's own sizing verdict for the requested GPU shape:
|
|
934
|
+
* recommended microbatch/accumulation/learning rate, predicted peak memory,
|
|
935
|
+
* minimum GPU count, wall-clock estimate, and whether the per-device batch
|
|
936
|
+
* you asked for is predicted to fit. Absent when no advisor is deployed.
|
|
937
|
+
*/
|
|
938
|
+
advisor?: TrainingAdvisorVerdict;
|
|
939
|
+
}
|
|
940
|
+
/** Parameters for `training.recommend()` (GET /api/training/recommend). */
|
|
941
|
+
export interface TrainingRecommendParams {
|
|
942
|
+
model: string;
|
|
943
|
+
modelRevision?: string;
|
|
944
|
+
integrationId?: string;
|
|
945
|
+
gpuType: string;
|
|
946
|
+
gpuCount?: number;
|
|
947
|
+
adapter?: 'full' | 'lora' | 'qlora';
|
|
948
|
+
method?: 'sft' | 'cpt' | 'pt';
|
|
949
|
+
maxLength?: number;
|
|
950
|
+
epochs?: number;
|
|
951
|
+
maxSteps?: number;
|
|
952
|
+
/** Your own microbatch, to be judged against the model. */
|
|
953
|
+
perDeviceTrainBatchSize?: number;
|
|
954
|
+
gradientAccumulationSteps?: number;
|
|
955
|
+
/** Datasets the job will train on; their measured token statistics feed the sizing. */
|
|
956
|
+
datasetIds?: string[];
|
|
957
|
+
workspaceId?: string;
|
|
958
|
+
}
|
|
959
|
+
/** Basis of one recommended value: measured on hardware, derived through a stated model, or an argued default. */
|
|
960
|
+
export type TrainingAdvisorBasis = 'measured' | 'derived' | 'judgement';
|
|
961
|
+
export interface TrainingAdvisorJustification {
|
|
962
|
+
field: string;
|
|
963
|
+
value: string;
|
|
964
|
+
reason: string;
|
|
965
|
+
basis: TrainingAdvisorBasis;
|
|
966
|
+
}
|
|
967
|
+
export interface TrainingAdvisorRecommendation {
|
|
968
|
+
per_device_train_batch_size: number;
|
|
969
|
+
gradient_accumulation_steps: number;
|
|
970
|
+
global_batch_size: number;
|
|
971
|
+
activation_checkpoint: string;
|
|
972
|
+
compile: boolean;
|
|
973
|
+
learning_rate: number;
|
|
974
|
+
warmup_steps: number;
|
|
975
|
+
parallelism: {
|
|
976
|
+
dp_replicate: number;
|
|
977
|
+
dp_shard: number;
|
|
978
|
+
tp: number;
|
|
979
|
+
pp: number;
|
|
980
|
+
cp: number;
|
|
981
|
+
ep: number;
|
|
982
|
+
};
|
|
983
|
+
/** Predicted peak reserved GPU memory per device, GiB. */
|
|
984
|
+
predicted_peak_gb: number;
|
|
985
|
+
/** (median, worst) percent the prediction ran over measurement on the calibration rows. */
|
|
986
|
+
memory_band_percent: [number, number];
|
|
987
|
+
memory_class: string;
|
|
988
|
+
predicted_mfu: number;
|
|
989
|
+
supervised_tokens_per_step: number;
|
|
990
|
+
predicted_roughness: number;
|
|
991
|
+
tokens_per_second: number;
|
|
992
|
+
throughput_basis: 'measured' | 'derived';
|
|
993
|
+
wall_clock: {
|
|
994
|
+
steps_per_epoch: number;
|
|
995
|
+
total_steps: number;
|
|
996
|
+
training_hours: number;
|
|
997
|
+
startup_minutes_estimate: number;
|
|
998
|
+
basis: TrainingAdvisorBasis;
|
|
999
|
+
} | null;
|
|
1000
|
+
}
|
|
1001
|
+
export interface TrainingAdvisorUserShape {
|
|
1002
|
+
per_device_train_batch_size: number;
|
|
1003
|
+
fits: boolean;
|
|
1004
|
+
largest_fitting_batch?: number;
|
|
1005
|
+
predicted_peak_gb?: number;
|
|
1006
|
+
reason?: string;
|
|
1007
|
+
/** Same global batch, a microbatch that fits. Present only when `fits` is false. */
|
|
1008
|
+
suggested?: {
|
|
1009
|
+
per_device_train_batch_size: number;
|
|
1010
|
+
gradient_accumulation_steps: number;
|
|
1011
|
+
reason: string;
|
|
1012
|
+
};
|
|
1013
|
+
}
|
|
1014
|
+
/**
|
|
1015
|
+
* Verdict of the training advisor. `available: false` means no advisor is
|
|
1016
|
+
* deployed or it did not answer; nothing else is populated then.
|
|
1017
|
+
*/
|
|
1018
|
+
export interface TrainingAdvisorVerdict {
|
|
1019
|
+
available: boolean;
|
|
1020
|
+
reason?: string;
|
|
1021
|
+
fits?: boolean;
|
|
1022
|
+
/** Smallest GPU count of this type the job fits on; null when none up to the per-job cap. */
|
|
1023
|
+
min_gpu_count?: number | null;
|
|
1024
|
+
gpu?: {
|
|
1025
|
+
platform_type: string;
|
|
1026
|
+
sized_as: string;
|
|
1027
|
+
capacity_gb: number;
|
|
1028
|
+
capacity_measured: boolean;
|
|
1029
|
+
};
|
|
1030
|
+
model?: {
|
|
1031
|
+
params_total_b: number;
|
|
1032
|
+
params_active_b: number;
|
|
1033
|
+
is_moe: boolean;
|
|
1034
|
+
is_vlm: boolean;
|
|
1035
|
+
has_linear_attention: boolean;
|
|
1036
|
+
};
|
|
1037
|
+
dataset_measured?: boolean;
|
|
1038
|
+
seq_len?: number;
|
|
1039
|
+
gpu_count?: number;
|
|
1040
|
+
recommended?: TrainingAdvisorRecommendation;
|
|
1041
|
+
user_shape?: TrainingAdvisorUserShape;
|
|
1042
|
+
justifications?: TrainingAdvisorJustification[];
|
|
1043
|
+
warnings?: string[];
|
|
1044
|
+
dataset_stats_used?: {
|
|
1045
|
+
num_rows: number;
|
|
1046
|
+
avg_tokens_per_sample?: number;
|
|
1047
|
+
avg_supervised_tokens_per_sample?: number;
|
|
1048
|
+
has_images: boolean;
|
|
1049
|
+
estimated: boolean;
|
|
1050
|
+
};
|
|
1051
|
+
measured_history?: {
|
|
1052
|
+
tokens_per_second: number;
|
|
1053
|
+
memory_anchors: number;
|
|
1054
|
+
};
|
|
1055
|
+
model_id?: string;
|
|
1056
|
+
model_revision?: string;
|
|
900
1057
|
}
|
|
901
1058
|
/** One method, algorithm, or adapter reported by the pinned training engine. */
|
|
902
1059
|
export interface TrainingCapabilityChoice {
|
|
@@ -916,6 +1073,8 @@ export interface TrainingConfigFieldCapability {
|
|
|
916
1073
|
label: string;
|
|
917
1074
|
type: 'integer' | 'number' | 'boolean' | 'string' | 'string_array' | 'string_or_string_array' | 'object';
|
|
918
1075
|
default: unknown;
|
|
1076
|
+
default_by_adapter?: Record<string, number>;
|
|
1077
|
+
requires_step_horizon?: boolean;
|
|
919
1078
|
enabled: boolean;
|
|
920
1079
|
disabled_reason?: string;
|
|
921
1080
|
aliases?: string[];
|
|
@@ -1817,7 +1976,7 @@ export interface ApiKey {
|
|
|
1817
1976
|
* on purpose: a scope added to the catalog must not make an existing SDK build
|
|
1818
1977
|
* reject a key it just read back from the API.
|
|
1819
1978
|
*/
|
|
1820
|
-
export type ApiKeyScope = 'models:read' | 'datasets:read' | 'datasets:write' | 'training:read' | 'training:write' | 'billing:read' | 'deployments:read' | 'deployments:write' | 'analytics:read' | 'integrations:read' | 'integrations:write' | 'skills:read' | 'skills:write' | 'serverless' | (string & {});
|
|
1979
|
+
export type ApiKeyScope = 'models:read' | 'datasets:read' | 'datasets:write' | 'training:read' | 'training:write' | 'billing:read' | 'deployments:read' | 'deployments:write' | 'analytics:read' | 'integrations:read' | 'integrations:write' | 'skills:read' | 'skills:write' | 'loop:read' | 'loop:write' | 'serverless' | (string & {});
|
|
1821
1980
|
/** Result of introspecting an API key — shows what it can do. */
|
|
1822
1981
|
export interface ApiKeyIntrospection {
|
|
1823
1982
|
auth_type: 'api_key' | 'jwt';
|
|
@@ -1953,3 +2112,458 @@ export interface ChatCompletionUsage {
|
|
|
1953
2112
|
total_tokens: number;
|
|
1954
2113
|
prompt_tokens_details?: PromptTokensDetails;
|
|
1955
2114
|
}
|
|
2115
|
+
/** What a training set is shaped for. The three need different things. */
|
|
2116
|
+
export type LoopMethod = 'sft' | 'dpo' | 'grpo' | 'kto';
|
|
2117
|
+
/** Who judged an answer. */
|
|
2118
|
+
export type LoopSource = 'human' | 'verifier' | 'judge' | 'behavioural';
|
|
2119
|
+
/** The judgement itself. */
|
|
2120
|
+
/**
|
|
2121
|
+
* The judgement recorded on an answer.
|
|
2122
|
+
*
|
|
2123
|
+
* `accepted` / `rejected` are the thumbs (see `upvote` / `downvote`).
|
|
2124
|
+
* `edited` says the model was WRONG and carries the better answer.
|
|
2125
|
+
* `gold` records the reference answer for the question, making no claim about
|
|
2126
|
+
* whether the model was right — which is why it is separate from `edited`.
|
|
2127
|
+
*/
|
|
2128
|
+
export type LoopVerdict = 'accepted' | 'rejected' | 'edited' | 'scored' | 'gold';
|
|
2129
|
+
/** A function an assistant turn asked to invoke. `arguments` is JSON-encoded. */
|
|
2130
|
+
export interface LoopToolCall {
|
|
2131
|
+
id?: string;
|
|
2132
|
+
type?: string;
|
|
2133
|
+
function: {
|
|
2134
|
+
name: string;
|
|
2135
|
+
arguments: string;
|
|
2136
|
+
};
|
|
2137
|
+
}
|
|
2138
|
+
/** One conversational turn, in the chat-completions shape. */
|
|
2139
|
+
export interface LoopMessage {
|
|
2140
|
+
role: string;
|
|
2141
|
+
content?: string;
|
|
2142
|
+
tool_calls?: LoopToolCall[];
|
|
2143
|
+
tool_call_id?: string;
|
|
2144
|
+
name?: string;
|
|
2145
|
+
}
|
|
2146
|
+
export interface LoopCaptureParams {
|
|
2147
|
+
/** The source being recorded. Must be enabled first, or nothing is stored. */
|
|
2148
|
+
deployment_id: string;
|
|
2149
|
+
model: string;
|
|
2150
|
+
messages: LoopMessage[];
|
|
2151
|
+
completion?: string;
|
|
2152
|
+
/** What the answering turn invoked, if anything. */
|
|
2153
|
+
tool_calls?: LoopToolCall[];
|
|
2154
|
+
/** The tool schema the model was offered. Without it, tool training invents names. */
|
|
2155
|
+
tools?: unknown;
|
|
2156
|
+
model_version?: string;
|
|
2157
|
+
/** Ties multi-turn work together. */
|
|
2158
|
+
conversation_id?: string;
|
|
2159
|
+
/** Your own idempotency handle. Replaying it returns the same trace. */
|
|
2160
|
+
request_id?: string;
|
|
2161
|
+
prompt_tokens?: number;
|
|
2162
|
+
completion_tokens?: number;
|
|
2163
|
+
latency_ms?: number;
|
|
2164
|
+
metadata?: Record<string, unknown>;
|
|
2165
|
+
}
|
|
2166
|
+
export interface LoopCaptureResult {
|
|
2167
|
+
captured: boolean;
|
|
2168
|
+
/** Present when captured. Ours, never the id you sent. */
|
|
2169
|
+
trace_id?: string;
|
|
2170
|
+
/** Set when something was removed before the record was written. */
|
|
2171
|
+
redacted?: boolean;
|
|
2172
|
+
download_url?: string;
|
|
2173
|
+
/** Why nothing was stored, e.g. `capture_disabled`, `not_sampled`. */
|
|
2174
|
+
reason?: string;
|
|
2175
|
+
}
|
|
2176
|
+
export interface LoopImportParams {
|
|
2177
|
+
/**
|
|
2178
|
+
* Where this came from. Required, and becomes the id every imported
|
|
2179
|
+
* conversation is filed under, so the import stays sliceable later.
|
|
2180
|
+
*/
|
|
2181
|
+
source: string;
|
|
2182
|
+
/** The file, already parsed. At most 5000 rows per call. */
|
|
2183
|
+
rows: Record<string, unknown>[];
|
|
2184
|
+
/** Which model produced these, when the rows do not say per-row. */
|
|
2185
|
+
model?: string;
|
|
2186
|
+
/** Applied to every row: importing a dump is when somebody knows what it is. */
|
|
2187
|
+
labels?: string[];
|
|
2188
|
+
attributes?: Record<string, string>;
|
|
2189
|
+
}
|
|
2190
|
+
export interface LoopImportResult {
|
|
2191
|
+
imported: number;
|
|
2192
|
+
/**
|
|
2193
|
+
* How many arrived carrying a verdict. Reported apart from `imported`
|
|
2194
|
+
* because it is the difference between data you can train on and data
|
|
2195
|
+
* somebody still has to look at.
|
|
2196
|
+
*/
|
|
2197
|
+
reviewed: number;
|
|
2198
|
+
needs_review: number;
|
|
2199
|
+
refused: number;
|
|
2200
|
+
/** How many rows of each recognised shape, so a mis-shaped file is visible. */
|
|
2201
|
+
by_shape: Record<string, number>;
|
|
2202
|
+
refused_why: Record<string, number>;
|
|
2203
|
+
notes: string[];
|
|
2204
|
+
}
|
|
2205
|
+
export interface LoopSignal {
|
|
2206
|
+
id: string;
|
|
2207
|
+
trace_id: string;
|
|
2208
|
+
source: LoopSource;
|
|
2209
|
+
verdict: LoopVerdict;
|
|
2210
|
+
correction: string | null;
|
|
2211
|
+
score: number | null;
|
|
2212
|
+
ground_truth: string | null;
|
|
2213
|
+
reason: string | null;
|
|
2214
|
+
author: string | null;
|
|
2215
|
+
created_at: string;
|
|
2216
|
+
}
|
|
2217
|
+
export interface LoopSignalParams {
|
|
2218
|
+
verdict: LoopVerdict;
|
|
2219
|
+
source?: LoopSource;
|
|
2220
|
+
/** Required when verdict is `edited`. The answer the model should have given. */
|
|
2221
|
+
correction?: string;
|
|
2222
|
+
/** Required when verdict is `scored`. */
|
|
2223
|
+
score?: number;
|
|
2224
|
+
/** A value or fact the answer can be checked against. GRPO needs one. */
|
|
2225
|
+
ground_truth?: string;
|
|
2226
|
+
reason?: string;
|
|
2227
|
+
author?: string;
|
|
2228
|
+
metadata?: Record<string, unknown>;
|
|
2229
|
+
}
|
|
2230
|
+
export interface LoopTrace {
|
|
2231
|
+
id: string;
|
|
2232
|
+
workspace_id: string;
|
|
2233
|
+
deployment_id: string | null;
|
|
2234
|
+
model: string;
|
|
2235
|
+
model_version: string | null;
|
|
2236
|
+
messages: LoopMessage[];
|
|
2237
|
+
completion: string;
|
|
2238
|
+
tool_calls?: LoopToolCall[];
|
|
2239
|
+
tools?: unknown;
|
|
2240
|
+
conversation_id: string | null;
|
|
2241
|
+
request_id: string | null;
|
|
2242
|
+
prompt_tokens: number | null;
|
|
2243
|
+
completion_tokens: number | null;
|
|
2244
|
+
latency_ms: number | null;
|
|
2245
|
+
redacted_at: string | null;
|
|
2246
|
+
/** What the redaction pass removed, counted by rule. */
|
|
2247
|
+
redaction_report?: Record<string, number>;
|
|
2248
|
+
created_at: string;
|
|
2249
|
+
expires_at: string;
|
|
2250
|
+
signals?: LoopSignal[];
|
|
2251
|
+
}
|
|
2252
|
+
export interface LoopTraceListParams {
|
|
2253
|
+
deployment_id?: string;
|
|
2254
|
+
conversation_id?: string;
|
|
2255
|
+
from?: string;
|
|
2256
|
+
to?: string;
|
|
2257
|
+
/** Only conversations that already carry a verdict. */
|
|
2258
|
+
signalled?: boolean;
|
|
2259
|
+
/** One bare tag. A tag with children matches them too. */
|
|
2260
|
+
label?: string;
|
|
2261
|
+
/**
|
|
2262
|
+
* Named dimensions, ANDed together: `{ category: 'billing', language: 'es' }`.
|
|
2263
|
+
* Exact within their key — `source=support` does not match `team=support`.
|
|
2264
|
+
*/
|
|
2265
|
+
attributes?: Record<string, string>;
|
|
2266
|
+
/** Only the conversations nobody has described yet. */
|
|
2267
|
+
unlabelled?: boolean;
|
|
2268
|
+
limit?: number;
|
|
2269
|
+
offset?: number;
|
|
2270
|
+
}
|
|
2271
|
+
export interface LoopTraceListResponse {
|
|
2272
|
+
traces: LoopTrace[];
|
|
2273
|
+
total: number;
|
|
2274
|
+
limit: number;
|
|
2275
|
+
offset: number;
|
|
2276
|
+
}
|
|
2277
|
+
export interface LoopDatasetCreateParams {
|
|
2278
|
+
name: string;
|
|
2279
|
+
method: LoopMethod;
|
|
2280
|
+
deployment_id?: string;
|
|
2281
|
+
from?: string;
|
|
2282
|
+
to?: string;
|
|
2283
|
+
/** Narrow to one slice. A label with children selects them too. */
|
|
2284
|
+
label?: string;
|
|
2285
|
+
/** Take this many at random from what the filters matched. */
|
|
2286
|
+
sample?: number;
|
|
2287
|
+
/** Holds back your most recent work, not a random slice. 0-50. */
|
|
2288
|
+
holdout_percent?: number;
|
|
2289
|
+
max_items?: number;
|
|
2290
|
+
/**
|
|
2291
|
+
* Named dimensions, ANDed with each other and with `label`:
|
|
2292
|
+
* `{ category: 'billing', language: 'es' }`. This is what turns one captured
|
|
2293
|
+
* corpus into a different dataset for every task somebody trains for.
|
|
2294
|
+
*/
|
|
2295
|
+
attributes?: Record<string, string>;
|
|
2296
|
+
/** Only the conversations nobody has described yet. */
|
|
2297
|
+
unlabelled?: boolean;
|
|
2298
|
+
}
|
|
2299
|
+
export interface LoopDataset {
|
|
2300
|
+
id: string;
|
|
2301
|
+
workspace_id: string;
|
|
2302
|
+
name: string;
|
|
2303
|
+
method: LoopMethod;
|
|
2304
|
+
status: string;
|
|
2305
|
+
spec: Record<string, unknown>;
|
|
2306
|
+
item_count: number;
|
|
2307
|
+
considered_count: number;
|
|
2308
|
+
/** Why rows were left out, by reason. */
|
|
2309
|
+
rejected_counts: Record<string, number>;
|
|
2310
|
+
holdout_count: number;
|
|
2311
|
+
holdout_cutoff: string | null;
|
|
2312
|
+
created_by: string | null;
|
|
2313
|
+
created_at: string;
|
|
2314
|
+
completed_at: string | null;
|
|
2315
|
+
download_url: string;
|
|
2316
|
+
}
|
|
2317
|
+
export interface LoopDatasetListParams {
|
|
2318
|
+
method?: LoopMethod;
|
|
2319
|
+
limit?: number;
|
|
2320
|
+
offset?: number;
|
|
2321
|
+
}
|
|
2322
|
+
export interface LoopDatasetItem {
|
|
2323
|
+
id: number;
|
|
2324
|
+
trace_id: string;
|
|
2325
|
+
split: 'train' | 'holdout';
|
|
2326
|
+
payload: Record<string, unknown>;
|
|
2327
|
+
dedup_key: string;
|
|
2328
|
+
created_at: string;
|
|
2329
|
+
/** The conversation this row was built from. */
|
|
2330
|
+
trace_url: string;
|
|
2331
|
+
}
|
|
2332
|
+
export interface LoopConfig {
|
|
2333
|
+
deployment_id: string;
|
|
2334
|
+
workspace_id: string;
|
|
2335
|
+
enabled: boolean;
|
|
2336
|
+
retention_days: number;
|
|
2337
|
+
sample_rate: number;
|
|
2338
|
+
enabled_by: string | null;
|
|
2339
|
+
enabled_at: string | null;
|
|
2340
|
+
updated_at: string;
|
|
2341
|
+
}
|
|
2342
|
+
export interface LoopConfigParams {
|
|
2343
|
+
enabled: boolean;
|
|
2344
|
+
retention_days?: number;
|
|
2345
|
+
/** Deterministic per conversation, so turns are never split. 0 < rate <= 1. */
|
|
2346
|
+
sample_rate?: number;
|
|
2347
|
+
}
|
|
2348
|
+
/** One thing a judge scores, separately from the others. */
|
|
2349
|
+
export interface LoopJudgeDimension {
|
|
2350
|
+
key: string;
|
|
2351
|
+
description?: string;
|
|
2352
|
+
}
|
|
2353
|
+
/** Which slice of the corpus a judge is responsible for. */
|
|
2354
|
+
export interface LoopJudgeSelection {
|
|
2355
|
+
label?: string;
|
|
2356
|
+
deployment_id?: string;
|
|
2357
|
+
from?: string;
|
|
2358
|
+
to?: string;
|
|
2359
|
+
sample?: number;
|
|
2360
|
+
/** Skip conversations that already carry a judge verdict. */
|
|
2361
|
+
only_unscored?: boolean;
|
|
2362
|
+
}
|
|
2363
|
+
export interface LoopJudge {
|
|
2364
|
+
id: string;
|
|
2365
|
+
workspace_id: string;
|
|
2366
|
+
name: string;
|
|
2367
|
+
instructions: string;
|
|
2368
|
+
dimensions: LoopJudgeDimension[];
|
|
2369
|
+
selection: LoopJudgeSelection;
|
|
2370
|
+
model: string | null;
|
|
2371
|
+
write_gold: boolean;
|
|
2372
|
+
enabled: boolean;
|
|
2373
|
+
created_by: string | null;
|
|
2374
|
+
created_at: string;
|
|
2375
|
+
updated_at: string;
|
|
2376
|
+
}
|
|
2377
|
+
export interface LoopJudgeParams {
|
|
2378
|
+
name: string;
|
|
2379
|
+
instructions: string;
|
|
2380
|
+
dimensions: LoopJudgeDimension[];
|
|
2381
|
+
selection?: LoopJudgeSelection;
|
|
2382
|
+
model?: string;
|
|
2383
|
+
/**
|
|
2384
|
+
* Let this judge write the answer that should have been given, not just a
|
|
2385
|
+
* score. Off by default: a reference answer written by a model and then
|
|
2386
|
+
* trained on is distillation, which is a decision you make deliberately.
|
|
2387
|
+
*/
|
|
2388
|
+
write_gold?: boolean;
|
|
2389
|
+
enabled?: boolean;
|
|
2390
|
+
}
|
|
2391
|
+
/** One pass of one rubric over one slice. */
|
|
2392
|
+
export interface LoopJudgeRun {
|
|
2393
|
+
id: string;
|
|
2394
|
+
judge_id: string;
|
|
2395
|
+
judge_name?: string;
|
|
2396
|
+
status: 'open' | 'done';
|
|
2397
|
+
instructions: string;
|
|
2398
|
+
dimensions: LoopJudgeDimension[];
|
|
2399
|
+
model: string | null;
|
|
2400
|
+
selected: number;
|
|
2401
|
+
scored: number;
|
|
2402
|
+
failed: number;
|
|
2403
|
+
created_by: string | null;
|
|
2404
|
+
created_at: string;
|
|
2405
|
+
finished_at: string | null;
|
|
2406
|
+
}
|
|
2407
|
+
/**
|
|
2408
|
+
* One conversation to score, already rendered into the prompt to send.
|
|
2409
|
+
*
|
|
2410
|
+
* Send `prompt` as-is. Assembling it yourself is how two callers end up giving
|
|
2411
|
+
* the same rubric different instructions, and two judges given different
|
|
2412
|
+
* instructions are not one judge.
|
|
2413
|
+
*/
|
|
2414
|
+
export interface LoopJudgeWorkItem {
|
|
2415
|
+
trace_id: string;
|
|
2416
|
+
messages: Array<{
|
|
2417
|
+
role: string;
|
|
2418
|
+
content?: string;
|
|
2419
|
+
}>;
|
|
2420
|
+
answer: string;
|
|
2421
|
+
prompt: string;
|
|
2422
|
+
}
|
|
2423
|
+
export interface LoopJudgeWork {
|
|
2424
|
+
run: LoopJudgeRun;
|
|
2425
|
+
items: LoopJudgeWorkItem[];
|
|
2426
|
+
remaining: number;
|
|
2427
|
+
}
|
|
2428
|
+
/** One scored conversation going back. */
|
|
2429
|
+
export interface LoopJudgeVerdict {
|
|
2430
|
+
trace_id: string;
|
|
2431
|
+
/** One score per dimension the rubric asked for, each between 0 and 1. */
|
|
2432
|
+
scores?: Record<string, number>;
|
|
2433
|
+
/** Your own combination of them. Left out, the plain mean is used. */
|
|
2434
|
+
overall?: number;
|
|
2435
|
+
reason?: string;
|
|
2436
|
+
/** Only stored if the judge was created with `write_gold`. */
|
|
2437
|
+
gold?: string;
|
|
2438
|
+
/** Mark an item you could not score, instead of dropping it silently. */
|
|
2439
|
+
error?: string;
|
|
2440
|
+
}
|
|
2441
|
+
export interface LoopJudgeVerdictResult {
|
|
2442
|
+
run: LoopJudgeRun;
|
|
2443
|
+
recorded: number;
|
|
2444
|
+
failed: number;
|
|
2445
|
+
rejected: Array<{
|
|
2446
|
+
trace_id: string;
|
|
2447
|
+
error: string;
|
|
2448
|
+
}>;
|
|
2449
|
+
}
|
|
2450
|
+
export interface LoopStats {
|
|
2451
|
+
traces: number;
|
|
2452
|
+
signalled_traces: number;
|
|
2453
|
+
signals: number;
|
|
2454
|
+
by_verdict: Record<string, number>;
|
|
2455
|
+
by_source: Record<string, number>;
|
|
2456
|
+
datasets: number;
|
|
2457
|
+
/** Upper bounds: deduplication runs when a set is built. */
|
|
2458
|
+
ready: Record<LoopMethod, number>;
|
|
2459
|
+
}
|
|
2460
|
+
export interface LoopCandidateParams {
|
|
2461
|
+
completion?: string;
|
|
2462
|
+
/** An alternative that is itself a tool call. */
|
|
2463
|
+
tool_calls?: LoopToolCall[];
|
|
2464
|
+
/** Which model produced this alternative. The teacher, when distilling. */
|
|
2465
|
+
model?: string;
|
|
2466
|
+
model_version?: string;
|
|
2467
|
+
/** Unscored alternatives cannot pair -- a missing score is not a low one. */
|
|
2468
|
+
score?: number;
|
|
2469
|
+
/** Required alongside a score: human | verifier | judge | behavioural. */
|
|
2470
|
+
score_source?: LoopSource;
|
|
2471
|
+
reason?: string;
|
|
2472
|
+
metadata?: Record<string, unknown>;
|
|
2473
|
+
}
|
|
2474
|
+
export interface LoopCandidate {
|
|
2475
|
+
id: string;
|
|
2476
|
+
trace_id: string;
|
|
2477
|
+
model: string | null;
|
|
2478
|
+
model_version: string | null;
|
|
2479
|
+
completion: string;
|
|
2480
|
+
tool_calls?: LoopToolCall[];
|
|
2481
|
+
score: number | null;
|
|
2482
|
+
score_source: LoopSource | null;
|
|
2483
|
+
reason: string | null;
|
|
2484
|
+
metadata: Record<string, unknown>;
|
|
2485
|
+
created_at: string;
|
|
2486
|
+
}
|
|
2487
|
+
/** The deterministic checks a grader can perform. */
|
|
2488
|
+
export type LoopGraderKind = 'exact_match' | 'contains' | 'regex' | 'json_valid' | 'numeric' | 'tool_called';
|
|
2489
|
+
export interface LoopGraderConfig {
|
|
2490
|
+
expected?: string;
|
|
2491
|
+
/** Every phrase that must appear. */
|
|
2492
|
+
required?: string[];
|
|
2493
|
+
/** Phrases that must not. On their own these are passed by silence. */
|
|
2494
|
+
forbidden?: string[];
|
|
2495
|
+
pattern?: string;
|
|
2496
|
+
/** Keys the answer must carry, for `json_valid`. */
|
|
2497
|
+
keys?: string[];
|
|
2498
|
+
/** How far from `expected` still counts, for `numeric`. */
|
|
2499
|
+
tolerance?: number;
|
|
2500
|
+
/** The function that must have been called, for `tool_called`. */
|
|
2501
|
+
function?: string;
|
|
2502
|
+
case_sensitive?: boolean;
|
|
2503
|
+
}
|
|
2504
|
+
export interface LoopGraderParams {
|
|
2505
|
+
name: string;
|
|
2506
|
+
kind: LoopGraderKind;
|
|
2507
|
+
config: LoopGraderConfig;
|
|
2508
|
+
/** The multiplier. Twice as important, twice the weight. Must exceed zero. */
|
|
2509
|
+
weight?: number;
|
|
2510
|
+
enabled?: boolean;
|
|
2511
|
+
/** Limit the rule to one capture source. Omit to apply it everywhere. */
|
|
2512
|
+
deployment_id?: string;
|
|
2513
|
+
/**
|
|
2514
|
+
* Aim the rule at one slice of the corpus. A master group covers its
|
|
2515
|
+
* children. Outside that slice the rule is NOT APPLIED, rather than failed,
|
|
2516
|
+
* so it stays out of the score entirely. Omit to apply it everywhere.
|
|
2517
|
+
*/
|
|
2518
|
+
label?: string;
|
|
2519
|
+
}
|
|
2520
|
+
export interface LoopGrader extends LoopGraderParams {
|
|
2521
|
+
id: string;
|
|
2522
|
+
workspace_id: string;
|
|
2523
|
+
weight: number;
|
|
2524
|
+
enabled: boolean;
|
|
2525
|
+
created_by: string | null;
|
|
2526
|
+
created_at: string;
|
|
2527
|
+
updated_at: string;
|
|
2528
|
+
}
|
|
2529
|
+
export interface LoopGraderResult {
|
|
2530
|
+
grader_id: string;
|
|
2531
|
+
name: string;
|
|
2532
|
+
kind: string;
|
|
2533
|
+
weight: number;
|
|
2534
|
+
passed: boolean;
|
|
2535
|
+
score: number;
|
|
2536
|
+
/** What happened, in words -- "failed" alone sends you to read the rule. */
|
|
2537
|
+
detail: string;
|
|
2538
|
+
/** The rule had no opinion. Excluded from the score rather than counted 0. */
|
|
2539
|
+
skipped: boolean;
|
|
2540
|
+
}
|
|
2541
|
+
export interface LoopGradeReport {
|
|
2542
|
+
/** Weighted mean over the rules that applied, 0 to 1. */
|
|
2543
|
+
score: number;
|
|
2544
|
+
results: LoopGraderResult[];
|
|
2545
|
+
/** The denominator. 1.0 from one rule is not 1.0 from six. */
|
|
2546
|
+
applied: number;
|
|
2547
|
+
skipped: number;
|
|
2548
|
+
}
|
|
2549
|
+
export interface LoopGradeResult {
|
|
2550
|
+
report: LoopGradeReport;
|
|
2551
|
+
/** Null when no rule applied, because no verdict was invented. */
|
|
2552
|
+
signal_id: string | null;
|
|
2553
|
+
candidates_graded: number;
|
|
2554
|
+
}
|
|
2555
|
+
export interface LoopLabel {
|
|
2556
|
+
label: string;
|
|
2557
|
+
/** The master group this label belongs to, if any. */
|
|
2558
|
+
parent: string | null;
|
|
2559
|
+
created_by: string | null;
|
|
2560
|
+
created_at: string;
|
|
2561
|
+
/** The dimension this value belongs to. `tag` for a bare name. */
|
|
2562
|
+
key: string;
|
|
2563
|
+
}
|
|
2564
|
+
export interface LoopLabelCount {
|
|
2565
|
+
label: string;
|
|
2566
|
+
parent: string | null;
|
|
2567
|
+
traces: number;
|
|
2568
|
+
key: string;
|
|
2569
|
+
}
|