runbios-sdk 0.2.1-rc.98 → 0.2.2-dev.171
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +32 -2
- package/dist/client.d.ts +48 -1
- package/dist/client.js +64 -1
- package/dist/index.d.ts +15 -3
- package/dist/index.js +16 -2
- package/dist/resources/datasets.d.ts +8 -4
- package/dist/resources/datasets.js +15 -4
- package/dist/resources/gpu.js +7 -0
- package/dist/resources/inference.d.ts +33 -2
- package/dist/resources/inference.js +46 -1
- package/dist/resources/integrations.d.ts +21 -0
- package/dist/resources/integrations.js +20 -0
- package/dist/resources/loop.d.ts +850 -0
- package/dist/resources/loop.js +1189 -0
- package/dist/resources/training.d.ts +20 -8
- package/dist/resources/training.js +52 -9
- package/dist/types.d.ts +1980 -6
- package/dist/types.js +11 -1
- package/package.json +2 -2
|
@@ -1,5 +1,5 @@
|
|
|
1
1
|
import type { HttpClient } from '../client.js';
|
|
2
|
-
import type { TrainingCreateParams, TrainingListParams, TrainingJob, TrainingListResponse, TrainingMetrics, TrainingCheckpoint, TrainingLogs, TrainingStopResponse, TrainingResumeResponse, TrainingPreflightResponse, TrainingCapabilities } from '../types.js';
|
|
2
|
+
import type { TrainingCreateParams, TrainingListParams, TrainingJob, TrainingListResponse, TrainingMetrics, TrainingCheckpoint, TrainingLogs, TrainingStopResponse, TrainingResumeResponse, TrainingPreflightResponse, TrainingRecommendParams, TrainingAdvisorVerdict, TrainingCapabilities } from '../types.js';
|
|
3
3
|
/**
|
|
4
4
|
* Create, monitor, and manage fine-tuning training jobs.
|
|
5
5
|
*/
|
|
@@ -12,8 +12,11 @@ export declare class Training {
|
|
|
12
12
|
/**
|
|
13
13
|
* Create a new training job (book-before-reveal).
|
|
14
14
|
*
|
|
15
|
-
*
|
|
16
|
-
*
|
|
15
|
+
* Choose method, adapter, datasets and the complete hyperparameter config
|
|
16
|
+
* first, call {@link preflight}, then choose one returned GPU configuration
|
|
17
|
+
* last. Its minimum is derived from every earlier choice. This call blocks
|
|
18
|
+
* while that placement is booked (~40s typical). A training id and the
|
|
19
|
+
* "training started" email exist only once a real machine is
|
|
17
20
|
* secured, so the returned `status` is one of:
|
|
18
21
|
*
|
|
19
22
|
* - `"booked"` — a GPU was secured (booked == secured); the job then
|
|
@@ -25,11 +28,12 @@ export declare class Training {
|
|
|
25
28
|
* (`queueIfUnavailable: true`); waits for stock at zero charge and books
|
|
26
29
|
* via the same path.
|
|
27
30
|
*
|
|
28
|
-
* If
|
|
29
|
-
*
|
|
30
|
-
*
|
|
31
|
-
*
|
|
32
|
-
*
|
|
31
|
+
* If booking fails, this rejects with `CAPACITY_UNAVAILABLE` (HTTP 409)
|
|
32
|
+
* carrying fresh qualifying alternatives — no phantom job and no charge.
|
|
33
|
+
* Keep every earlier field unchanged, select one alternative and retry. The
|
|
34
|
+
* rejected type is never recommended back. Queue consent is appropriate only
|
|
35
|
+
* when the alternatives list is empty; waiting is unbilled. The SDK never
|
|
36
|
+
* auto-substitutes a GPU; a transient 503 is a retry, never a capacity verdict.
|
|
33
37
|
*
|
|
34
38
|
* @example
|
|
35
39
|
* ```ts
|
|
@@ -51,6 +55,14 @@ export declare class Training {
|
|
|
51
55
|
create(params: TrainingCreateParams): Promise<TrainingJob>;
|
|
52
56
|
/** Validate and canonicalize a training request without creating or billing a job. */
|
|
53
57
|
preflight(params: TrainingCreateParams): Promise<TrainingPreflightResponse>;
|
|
58
|
+
/**
|
|
59
|
+
* Ask the trainer's own sizing model what to run: per-device batch,
|
|
60
|
+
* accumulation, learning rate, warmup, predicted peak memory, minimum GPU
|
|
61
|
+
* count and a wall-clock estimate for this model on this GPU type, with the
|
|
62
|
+
* basis of every number. Side-effect free. When `available` is false no
|
|
63
|
+
* advisor is deployed and the other fields are absent -- nothing is guessed.
|
|
64
|
+
*/
|
|
65
|
+
recommend(params: TrainingRecommendParams): Promise<TrainingAdvisorVerdict>;
|
|
54
66
|
/** Return one server-driven page with pagination metadata. */
|
|
55
67
|
listPage(params?: TrainingListParams): Promise<TrainingListResponse>;
|
|
56
68
|
/**
|
|
@@ -67,10 +67,12 @@ function buildTrainingRequest(params) {
|
|
|
67
67
|
body.integration_id = params.integrationId;
|
|
68
68
|
if (params.networkVolumeId !== undefined)
|
|
69
69
|
body.network_volume_id = params.networkVolumeId;
|
|
70
|
-
if (params.cacheDataset !== undefined)
|
|
71
|
-
body.cache_dataset = params.cacheDataset;
|
|
72
70
|
if (params.datasetSampleLimits !== undefined)
|
|
73
71
|
body.dataset_sample_limits = params.datasetSampleLimits;
|
|
72
|
+
if (params.datasetSampling !== undefined)
|
|
73
|
+
body.dataset_sampling = params.datasetSampling;
|
|
74
|
+
if (params.datasetSamplingStrategy !== undefined)
|
|
75
|
+
body.dataset_sampling_strategy = params.datasetSamplingStrategy;
|
|
74
76
|
if (params.datasetMixing !== undefined)
|
|
75
77
|
body.dataset_mixing = params.datasetMixing;
|
|
76
78
|
if (params.mixing !== undefined)
|
|
@@ -78,6 +80,8 @@ function buildTrainingRequest(params) {
|
|
|
78
80
|
const config = {};
|
|
79
81
|
if (params.epochs !== undefined)
|
|
80
82
|
config.num_train_epochs = params.epochs;
|
|
83
|
+
if (params.maxSteps !== undefined)
|
|
84
|
+
config.max_steps = params.maxSteps;
|
|
81
85
|
if (params.batchSize !== undefined)
|
|
82
86
|
config.per_device_train_batch_size = params.batchSize;
|
|
83
87
|
if (params.gradientAccumulation !== undefined)
|
|
@@ -183,8 +187,11 @@ export class Training {
|
|
|
183
187
|
/**
|
|
184
188
|
* Create a new training job (book-before-reveal).
|
|
185
189
|
*
|
|
186
|
-
*
|
|
187
|
-
*
|
|
190
|
+
* Choose method, adapter, datasets and the complete hyperparameter config
|
|
191
|
+
* first, call {@link preflight}, then choose one returned GPU configuration
|
|
192
|
+
* last. Its minimum is derived from every earlier choice. This call blocks
|
|
193
|
+
* while that placement is booked (~40s typical). A training id and the
|
|
194
|
+
* "training started" email exist only once a real machine is
|
|
188
195
|
* secured, so the returned `status` is one of:
|
|
189
196
|
*
|
|
190
197
|
* - `"booked"` — a GPU was secured (booked == secured); the job then
|
|
@@ -196,11 +203,12 @@ export class Training {
|
|
|
196
203
|
* (`queueIfUnavailable: true`); waits for stock at zero charge and books
|
|
197
204
|
* via the same path.
|
|
198
205
|
*
|
|
199
|
-
* If
|
|
200
|
-
*
|
|
201
|
-
*
|
|
202
|
-
*
|
|
203
|
-
*
|
|
206
|
+
* If booking fails, this rejects with `CAPACITY_UNAVAILABLE` (HTTP 409)
|
|
207
|
+
* carrying fresh qualifying alternatives — no phantom job and no charge.
|
|
208
|
+
* Keep every earlier field unchanged, select one alternative and retry. The
|
|
209
|
+
* rejected type is never recommended back. Queue consent is appropriate only
|
|
210
|
+
* when the alternatives list is empty; waiting is unbilled. The SDK never
|
|
211
|
+
* auto-substitutes a GPU; a transient 503 is a retry, never a capacity verdict.
|
|
204
212
|
*
|
|
205
213
|
* @example
|
|
206
214
|
* ```ts
|
|
@@ -240,6 +248,41 @@ export class Training {
|
|
|
240
248
|
const { body } = buildTrainingRequest(params);
|
|
241
249
|
return this._http.fetchPost('/api/training/preflight', body);
|
|
242
250
|
}
|
|
251
|
+
/**
|
|
252
|
+
* Ask the trainer's own sizing model what to run: per-device batch,
|
|
253
|
+
* accumulation, learning rate, warmup, predicted peak memory, minimum GPU
|
|
254
|
+
* count and a wall-clock estimate for this model on this GPU type, with the
|
|
255
|
+
* basis of every number. Side-effect free. When `available` is false no
|
|
256
|
+
* advisor is deployed and the other fields are absent -- nothing is guessed.
|
|
257
|
+
*/
|
|
258
|
+
async recommend(params) {
|
|
259
|
+
const q = new URLSearchParams({ model_id: params.model, gpu_type: params.gpuType });
|
|
260
|
+
if (params.modelRevision)
|
|
261
|
+
q.set('model_revision', params.modelRevision);
|
|
262
|
+
if (params.integrationId)
|
|
263
|
+
q.set('integration_id', params.integrationId);
|
|
264
|
+
if (params.gpuCount !== undefined)
|
|
265
|
+
q.set('gpu_count', String(params.gpuCount));
|
|
266
|
+
if (params.adapter)
|
|
267
|
+
q.set('train_type', params.adapter);
|
|
268
|
+
if (params.method)
|
|
269
|
+
q.set('method', params.method === 'cpt' ? 'pt' : params.method);
|
|
270
|
+
if (params.maxLength !== undefined)
|
|
271
|
+
q.set('max_length', String(params.maxLength));
|
|
272
|
+
if (params.epochs !== undefined)
|
|
273
|
+
q.set('num_train_epochs', String(params.epochs));
|
|
274
|
+
if (params.maxSteps !== undefined)
|
|
275
|
+
q.set('max_steps', String(params.maxSteps));
|
|
276
|
+
if (params.perDeviceTrainBatchSize !== undefined)
|
|
277
|
+
q.set('per_device_train_batch_size', String(params.perDeviceTrainBatchSize));
|
|
278
|
+
if (params.gradientAccumulationSteps !== undefined)
|
|
279
|
+
q.set('gradient_accumulation_steps', String(params.gradientAccumulationSteps));
|
|
280
|
+
if (params.datasetIds?.length)
|
|
281
|
+
q.set('dataset_ids', params.datasetIds.join(','));
|
|
282
|
+
if (params.workspaceId)
|
|
283
|
+
q.set('workspace_id', params.workspaceId);
|
|
284
|
+
return this._http.fetchGet(`/api/training/recommend?${q}`);
|
|
285
|
+
}
|
|
243
286
|
/** Return one server-driven page with pagination metadata. */
|
|
244
287
|
async listPage(params = {}) {
|
|
245
288
|
const q = new URLSearchParams();
|