runbios-sdk 0.2.1-dev.116 → 0.2.1-dev.120

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/dist/index.d.ts CHANGED
@@ -36,7 +36,7 @@ import { Loop } from './resources/loop.js';
36
36
  * SDK version. Sent as part of the User-Agent header.
37
37
  * Must match package.json "version" -- enforced by a contract test.
38
38
  */
39
- export declare const VERSION = "0.2.1-dev.116";
39
+ export declare const VERSION = "0.2.1-dev.120";
40
40
  export declare class RunBiOS {
41
41
  /** Search models, fetch configs, check adapter compatibility. */
42
42
  readonly models: Models;
@@ -75,4 +75,4 @@ export { Training } from './resources/training.js';
75
75
  export { Wallet } from './resources/wallet.js';
76
76
  export { GPU, type GPURecommendation } from './resources/gpu.js';
77
77
  export { Inference, validateChatRequest, parseSSE, buildInferenceRequest, contextSizingBasis, type ContextSizingBasis, CONTEXT_DEFAULT_CEILING, CONTEXT_EDITABLE_FLOOR, type ChatMessage, type FunctionTool, type ChatCompletionParams, type ChatCompletionResponse, type ChatCompletionChunk, } from './resources/inference.js';
78
- export type { BiOSConfig, PaginatedResponse, ApiErrorBody, AvailableGpuAlternative, CapacityMinimumRequirement, InferenceBookingAccepted, Model, ModelSearchParams, ModelSearchResponse, ModelDetailResponse, ModelConfig, Dataset, DatasetListParams, DatasetListResponse, DatasetUploadParams, DatasetPreview, DatasetPreviewParams, DatasetImportHFParams, DatasetRegisterHFParams, DatasetHubSearchParams, DatasetHubPreviewParams, DatasetValidation, DatasetFormatVariant, DatasetFormatSpec, DatasetFormatSpecs, DatasetStorageUsage, TrainingMethod, RLHFAlgorithm, AdapterType, TrainingJobStatus, TrainingStatusFilter, TrainingCreateParams, TrainingListParams, TrainingListResponse, TrainingJob, TrainingMetrics, TrainingResourceMetrics, TrainingDeviceMetrics, MetricPoint, MetricGraphConfig, TrainingCheckpoint, TrainingLogs, TrainingLogEntry, TrainingStopResponse, TrainingResumeResponse, CanonicalTrainingRequest, TrainingPreflightDataset, TrainingPreflightWarning, TrainingPreflightResponse, TrainingCapabilityChoice, TrainingConfigFieldCapability, TrainingCapabilities, GPUChoice, WalletBalance, Transaction, TransactionListResponse, TransactionListParams, GPUInfo, GPUPricingResponse, GPUOptionsParams, GPUOption, GPUOptionSuggestion, GPUOptionsResponse, InferenceStatus, InferenceCreateParams, InferenceUpdateParams, InferenceDeployment, InferenceDeploymentSummary, InferenceCreateResponse, InferenceListResponse, InferenceUpdateResponse, InferenceLifecycleResponse, InferenceDeleteResponse, InferenceGPUOptionMarket, InferenceGPUOption, InferenceGPUAlternative, InferenceGPUOptionsParams, InferenceGPUOptionsResponse, InferenceCanonicalRequest, InferencePreflightWarning, InferencePreflightResponse, AdapterCompatibility, AdapterCompatibilityResponse, AdapterCompatibilityParams, SupportedArchitecture, ArchitectureScope, SupportedArchitecturesResponse, ApiKey, ApiKeyScope, ApiKeyIntrospection, Organization, OrgMember, OrgInvite, Workspace, Integration, IntegrationCreateParams, StorageUsage, StorageObject, PromptTokensDetails, ChatCompletionUsage, } from './types.js';
78
+ export type { BiOSConfig, PaginatedResponse, ApiErrorBody, AvailableGpuAlternative, CapacityMinimumRequirement, InferenceBookingAccepted, Model, ModelSearchParams, ModelSearchResponse, ModelDetailResponse, ModelConfig, Dataset, DatasetListParams, DatasetListResponse, DatasetUploadParams, DatasetPreview, DatasetPreviewParams, DatasetImportHFParams, DatasetRegisterHFParams, DatasetHubSearchParams, DatasetHubPreviewParams, DatasetValidation, DatasetFormatVariant, DatasetFormatSpec, DatasetFormatSpecs, DatasetStorageUsage, TrainingMethod, RLHFAlgorithm, AdapterType, TrainingJobStatus, TrainingStatusFilter, TrainingCreateParams, TrainingListParams, TrainingListResponse, TrainingJob, TrainingMetrics, TrainingResourceMetrics, TrainingDeviceMetrics, MetricPoint, MetricGraphConfig, TrainingCheckpoint, TrainingLogs, TrainingLogEntry, TrainingStopResponse, TrainingResumeResponse, CanonicalTrainingRequest, TrainingPreflightDataset, TrainingPreflightWarning, TrainingPreflightResponse, TrainingCapabilityChoice, TrainingConfigFieldCapability, TrainingCapabilities, GPUChoice, WalletBalance, Transaction, TransactionListResponse, TransactionListParams, GPUInfo, GPUPricingResponse, GPUOptionsParams, TrainingRecommendParams, TrainingAdvisorVerdict, TrainingAdvisorRecommendation, TrainingAdvisorUserShape, TrainingAdvisorJustification, TrainingAdvisorBasis, GPUOption, GPUOptionSuggestion, GPUOptionsResponse, InferenceStatus, InferenceCreateParams, InferenceUpdateParams, InferenceDeployment, InferenceDeploymentSummary, InferenceCreateResponse, InferenceListResponse, InferenceUpdateResponse, InferenceLifecycleResponse, InferenceDeleteResponse, InferenceGPUOptionMarket, InferenceGPUOption, InferenceGPUAlternative, InferenceGPUOptionsParams, InferenceGPUOptionsResponse, InferenceCanonicalRequest, InferencePreflightWarning, InferencePreflightResponse, AdapterCompatibility, AdapterCompatibilityResponse, AdapterCompatibilityParams, SupportedArchitecture, ArchitectureScope, SupportedArchitecturesResponse, ApiKey, ApiKeyScope, ApiKeyIntrospection, Organization, OrgMember, OrgInvite, Workspace, Integration, IntegrationCreateParams, StorageUsage, StorageObject, PromptTokensDetails, ChatCompletionUsage, } from './types.js';
package/dist/index.js CHANGED
@@ -36,7 +36,7 @@ import { Loop } from './resources/loop.js';
36
36
  * SDK version. Sent as part of the User-Agent header.
37
37
  * Must match package.json "version" -- enforced by a contract test.
38
38
  */
39
- export const VERSION = '0.2.1-dev.116';
39
+ export const VERSION = '0.2.1-dev.120';
40
40
  export class RunBiOS {
41
41
  /** Search models, fetch configs, check adapter compatibility. */
42
42
  models;
@@ -46,6 +46,13 @@ export class GPU {
46
46
  if (params.modelActiveParamsB !== undefined) {
47
47
  q.set('model_active_params_b', String(params.modelActiveParamsB));
48
48
  }
49
+ if (params.maxLength !== undefined)
50
+ q.set('max_length', String(params.maxLength));
51
+ if (params.effectiveBatch !== undefined)
52
+ q.set('effective_batch', String(params.effectiveBatch));
53
+ else if (params.perDeviceTrainBatchSize !== undefined) {
54
+ q.set('per_device_train_batch_size', String(params.perDeviceTrainBatchSize));
55
+ }
49
56
  return this._http.fetchGet(`/api/training/gpu-options?${q}`);
50
57
  }
51
58
  /**
@@ -1,5 +1,5 @@
1
1
  import type { HttpClient } from '../client.js';
2
- import type { TrainingCreateParams, TrainingListParams, TrainingJob, TrainingListResponse, TrainingMetrics, TrainingCheckpoint, TrainingLogs, TrainingStopResponse, TrainingResumeResponse, TrainingPreflightResponse, TrainingCapabilities } from '../types.js';
2
+ import type { TrainingCreateParams, TrainingListParams, TrainingJob, TrainingListResponse, TrainingMetrics, TrainingCheckpoint, TrainingLogs, TrainingStopResponse, TrainingResumeResponse, TrainingPreflightResponse, TrainingRecommendParams, TrainingAdvisorVerdict, TrainingCapabilities } from '../types.js';
3
3
  /**
4
4
  * Create, monitor, and manage fine-tuning training jobs.
5
5
  */
@@ -55,6 +55,14 @@ export declare class Training {
55
55
  create(params: TrainingCreateParams): Promise<TrainingJob>;
56
56
  /** Validate and canonicalize a training request without creating or billing a job. */
57
57
  preflight(params: TrainingCreateParams): Promise<TrainingPreflightResponse>;
58
+ /**
59
+ * Ask the trainer's own sizing model what to run: per-device batch,
60
+ * accumulation, learning rate, warmup, predicted peak memory, minimum GPU
61
+ * count and a wall-clock estimate for this model on this GPU type, with the
62
+ * basis of every number. Side-effect free. When `available` is false no
63
+ * advisor is deployed and the other fields are absent -- nothing is guessed.
64
+ */
65
+ recommend(params: TrainingRecommendParams): Promise<TrainingAdvisorVerdict>;
58
66
  /** Return one server-driven page with pagination metadata. */
59
67
  listPage(params?: TrainingListParams): Promise<TrainingListResponse>;
60
68
  /**
@@ -244,6 +244,41 @@ export class Training {
244
244
  const { body } = buildTrainingRequest(params);
245
245
  return this._http.fetchPost('/api/training/preflight', body);
246
246
  }
247
+ /**
248
+ * Ask the trainer's own sizing model what to run: per-device batch,
249
+ * accumulation, learning rate, warmup, predicted peak memory, minimum GPU
250
+ * count and a wall-clock estimate for this model on this GPU type, with the
251
+ * basis of every number. Side-effect free. When `available` is false no
252
+ * advisor is deployed and the other fields are absent -- nothing is guessed.
253
+ */
254
+ async recommend(params) {
255
+ const q = new URLSearchParams({ model_id: params.model, gpu_type: params.gpuType });
256
+ if (params.modelRevision)
257
+ q.set('model_revision', params.modelRevision);
258
+ if (params.integrationId)
259
+ q.set('integration_id', params.integrationId);
260
+ if (params.gpuCount !== undefined)
261
+ q.set('gpu_count', String(params.gpuCount));
262
+ if (params.adapter)
263
+ q.set('train_type', params.adapter);
264
+ if (params.method)
265
+ q.set('method', params.method === 'cpt' ? 'pt' : params.method);
266
+ if (params.maxLength !== undefined)
267
+ q.set('max_length', String(params.maxLength));
268
+ if (params.epochs !== undefined)
269
+ q.set('num_train_epochs', String(params.epochs));
270
+ if (params.maxSteps !== undefined)
271
+ q.set('max_steps', String(params.maxSteps));
272
+ if (params.perDeviceTrainBatchSize !== undefined)
273
+ q.set('per_device_train_batch_size', String(params.perDeviceTrainBatchSize));
274
+ if (params.gradientAccumulationSteps !== undefined)
275
+ q.set('gradient_accumulation_steps', String(params.gradientAccumulationSteps));
276
+ if (params.datasetIds?.length)
277
+ q.set('dataset_ids', params.datasetIds.join(','));
278
+ if (params.workspaceId)
279
+ q.set('workspace_id', params.workspaceId);
280
+ return this._http.fetchGet(`/api/training/recommend?${q}`);
281
+ }
247
282
  /** Return one server-driven page with pagination metadata. */
248
283
  async listPage(params = {}) {
249
284
  const q = new URLSearchParams();
package/dist/types.d.ts CHANGED
@@ -907,6 +907,8 @@ export interface TrainingPreflightDataset {
907
907
  export interface TrainingPreflightWarning {
908
908
  code: string;
909
909
  message: string;
910
+ /** Request field the warning is about (e.g. `per_device_train_batch_size`), when there is one. */
911
+ field?: string;
910
912
  }
911
913
  /** Side-effect-free validation/sizing result; this endpoint never creates or bills a job. */
912
914
  export interface TrainingPreflightResponse {
@@ -927,6 +929,131 @@ export interface TrainingPreflightResponse {
927
929
  queue_eligible: boolean;
928
930
  warnings: TrainingPreflightWarning[];
929
931
  checked_at: string;
932
+ /**
933
+ * The trainer image's own sizing verdict for the requested GPU shape:
934
+ * recommended microbatch/accumulation/learning rate, predicted peak memory,
935
+ * minimum GPU count, wall-clock estimate, and whether the per-device batch
936
+ * you asked for is predicted to fit. Absent when no advisor is deployed.
937
+ */
938
+ advisor?: TrainingAdvisorVerdict;
939
+ }
940
+ /** Parameters for `training.recommend()` (GET /api/training/recommend). */
941
+ export interface TrainingRecommendParams {
942
+ model: string;
943
+ modelRevision?: string;
944
+ integrationId?: string;
945
+ gpuType: string;
946
+ gpuCount?: number;
947
+ adapter?: 'full' | 'lora' | 'qlora';
948
+ method?: 'sft' | 'cpt' | 'pt';
949
+ maxLength?: number;
950
+ epochs?: number;
951
+ maxSteps?: number;
952
+ /** Your own microbatch, to be judged against the model. */
953
+ perDeviceTrainBatchSize?: number;
954
+ gradientAccumulationSteps?: number;
955
+ /** Datasets the job will train on; their measured token statistics feed the sizing. */
956
+ datasetIds?: string[];
957
+ workspaceId?: string;
958
+ }
959
+ /** Basis of one recommended value: measured on hardware, derived through a stated model, or an argued default. */
960
+ export type TrainingAdvisorBasis = 'measured' | 'derived' | 'judgement';
961
+ export interface TrainingAdvisorJustification {
962
+ field: string;
963
+ value: string;
964
+ reason: string;
965
+ basis: TrainingAdvisorBasis;
966
+ }
967
+ export interface TrainingAdvisorRecommendation {
968
+ per_device_train_batch_size: number;
969
+ gradient_accumulation_steps: number;
970
+ global_batch_size: number;
971
+ activation_checkpoint: string;
972
+ compile: boolean;
973
+ learning_rate: number;
974
+ warmup_steps: number;
975
+ parallelism: {
976
+ dp_replicate: number;
977
+ dp_shard: number;
978
+ tp: number;
979
+ pp: number;
980
+ cp: number;
981
+ ep: number;
982
+ };
983
+ /** Predicted peak reserved GPU memory per device, GiB. */
984
+ predicted_peak_gb: number;
985
+ /** (median, worst) percent the prediction ran over measurement on the calibration rows. */
986
+ memory_band_percent: [number, number];
987
+ memory_class: string;
988
+ predicted_mfu: number;
989
+ supervised_tokens_per_step: number;
990
+ predicted_roughness: number;
991
+ tokens_per_second: number;
992
+ throughput_basis: 'measured' | 'derived';
993
+ wall_clock: {
994
+ steps_per_epoch: number;
995
+ total_steps: number;
996
+ training_hours: number;
997
+ startup_minutes_estimate: number;
998
+ basis: TrainingAdvisorBasis;
999
+ } | null;
1000
+ }
1001
+ export interface TrainingAdvisorUserShape {
1002
+ per_device_train_batch_size: number;
1003
+ fits: boolean;
1004
+ largest_fitting_batch?: number;
1005
+ predicted_peak_gb?: number;
1006
+ reason?: string;
1007
+ /** Same global batch, a microbatch that fits. Present only when `fits` is false. */
1008
+ suggested?: {
1009
+ per_device_train_batch_size: number;
1010
+ gradient_accumulation_steps: number;
1011
+ reason: string;
1012
+ };
1013
+ }
1014
+ /**
1015
+ * Verdict of the training advisor. `available: false` means no advisor is
1016
+ * deployed or it did not answer; nothing else is populated then.
1017
+ */
1018
+ export interface TrainingAdvisorVerdict {
1019
+ available: boolean;
1020
+ reason?: string;
1021
+ fits?: boolean;
1022
+ /** Smallest GPU count of this type the job fits on; null when none up to the per-job cap. */
1023
+ min_gpu_count?: number | null;
1024
+ gpu?: {
1025
+ platform_type: string;
1026
+ sized_as: string;
1027
+ capacity_gb: number;
1028
+ capacity_measured: boolean;
1029
+ };
1030
+ model?: {
1031
+ params_total_b: number;
1032
+ params_active_b: number;
1033
+ is_moe: boolean;
1034
+ is_vlm: boolean;
1035
+ has_linear_attention: boolean;
1036
+ };
1037
+ dataset_measured?: boolean;
1038
+ seq_len?: number;
1039
+ gpu_count?: number;
1040
+ recommended?: TrainingAdvisorRecommendation;
1041
+ user_shape?: TrainingAdvisorUserShape;
1042
+ justifications?: TrainingAdvisorJustification[];
1043
+ warnings?: string[];
1044
+ dataset_stats_used?: {
1045
+ num_rows: number;
1046
+ avg_tokens_per_sample?: number;
1047
+ avg_supervised_tokens_per_sample?: number;
1048
+ has_images: boolean;
1049
+ estimated: boolean;
1050
+ };
1051
+ measured_history?: {
1052
+ tokens_per_second: number;
1053
+ memory_anchors: number;
1054
+ };
1055
+ model_id?: string;
1056
+ model_revision?: string;
930
1057
  }
931
1058
  /** One method, algorithm, or adapter reported by the pinned training engine. */
932
1059
  export interface TrainingCapabilityChoice {
@@ -1099,6 +1226,17 @@ export interface GPUOptionsParams {
1099
1226
  rlhfType?: RLHFAlgorithm | string;
1100
1227
  modelParamsB?: number;
1101
1228
  modelActiveParamsB?: number;
1229
+ /** Training sequence length the sizing should assume (tokens). */
1230
+ maxLength?: number;
1231
+ /**
1232
+ * Effective batch (samples per optimizer step) -- the quality decision made
1233
+ * before a GPU is chosen. Each option then answers with the per-device split
1234
+ * that runs it there (`recommended_micro_batch` x `recommended_grad_accum`
1235
+ * x `required_count`), and the minimum GPU count is sized at micro-batch 1.
1236
+ */
1237
+ effectiveBatch?: number;
1238
+ /** Per-device batch to size as typed instead (ignored when effectiveBatch is set). */
1239
+ perDeviceTrainBatchSize?: number;
1102
1240
  }
1103
1241
  /** One model-aware GPU option with live stock and total-price context. */
1104
1242
  export interface GPUOption {
@@ -1119,6 +1257,10 @@ export interface GPUOption {
1119
1257
  bookable: boolean;
1120
1258
  reason?: string;
1121
1259
  checked_at?: string;
1260
+ /** Present when the request stated an effective batch: its split on this card at required_count. */
1261
+ effective_batch?: number;
1262
+ recommended_micro_batch?: number;
1263
+ recommended_grad_accum?: number;
1122
1264
  }
1123
1265
  /** Actionable alternative returned when no GPU option is currently bookable. */
1124
1266
  export interface GPUOptionSuggestion {
@@ -1148,6 +1290,10 @@ export interface GPUOptionsResponse {
1148
1290
  storage_gb: number;
1149
1291
  price_per_hour_cents: number;
1150
1292
  total_price_per_hour_cents: number;
1293
+ /** Present when the request stated an effective batch. */
1294
+ effective_batch?: number;
1295
+ per_device_train_batch_size?: number;
1296
+ gradient_accumulation_steps?: number;
1151
1297
  };
1152
1298
  suggestions?: GPUOptionSuggestion[];
1153
1299
  }
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "runbios-sdk",
3
- "version": "0.2.1-dev.116",
3
+ "version": "0.2.1-dev.120",
4
4
  "description": "Official TypeScript SDK for the Run BiOS training and deployment platform API",
5
5
  "type": "module",
6
6
  "main": "./dist/index.js",