runbios-sdk 0.2.19 → 0.2.20-dev.283
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/dist/client.js +1 -1
- package/dist/index.d.ts +2 -2
- package/dist/index.js +1 -1
- package/dist/resources/inference.js +1 -1
- package/dist/resources/loop.d.ts +28 -24
- package/dist/resources/loop.js +28 -24
- package/dist/types.d.ts +157 -106
- package/package.json +1 -1
package/dist/client.js
CHANGED
|
@@ -344,7 +344,7 @@ export class HttpClient {
|
|
|
344
344
|
constructor(config) {
|
|
345
345
|
// Default host stays api.runbios.ai for now; cutover to api.runbios.ai is
|
|
346
346
|
// planned once its DNS exists.
|
|
347
|
-
this.baseUrl = (config.baseUrl || envBaseUrl() || 'https://api.runbios.ai').replace(/\/+$/, '');
|
|
347
|
+
this.baseUrl = (config.baseUrl || envBaseUrl() || 'https://api-dev.runbios.ai').replace(/\/+$/, '');
|
|
348
348
|
this.apiKey = config.apiKey ?? envApiKey();
|
|
349
349
|
this.accessToken = config.accessToken;
|
|
350
350
|
this.orgId = config.orgId;
|
package/dist/index.d.ts
CHANGED
|
@@ -36,7 +36,7 @@ import { Loop } from './resources/loop.js';
|
|
|
36
36
|
* SDK version. Sent as part of the User-Agent header.
|
|
37
37
|
* Must match package.json "version" -- enforced by a contract test.
|
|
38
38
|
*/
|
|
39
|
-
export declare const VERSION = "0.2.
|
|
39
|
+
export declare const VERSION = "0.2.20-dev.283";
|
|
40
40
|
export declare class RunBiOS {
|
|
41
41
|
/** Search models, fetch configs, check adapter compatibility. */
|
|
42
42
|
readonly models: Models;
|
|
@@ -75,7 +75,7 @@ export { Training } from './resources/training.js';
|
|
|
75
75
|
export { Wallet } from './resources/wallet.js';
|
|
76
76
|
export { GPU, type GPURecommendation } from './resources/gpu.js';
|
|
77
77
|
export { Inference, validateChatRequest, parseSSE, buildInferenceRequest, contextSizingBasis, type ContextSizingBasis, CONTEXT_DEFAULT_CEILING, CONTEXT_EDITABLE_FLOOR, type ChatMessage, type FunctionTool, type ChatCompletionParams, type ChatCompletionResponse, type ChatCompletionChunk, type InferenceVerbParams, type CompletionParams, type EmbeddingParams, type RerankParams, type AnthropicMessageParams, type ServerlessUsageWindow, type ServerlessTimeseriesMetric, type ServerlessWorkspaceLimits, type ServerlessUsageEnvelope, type ServerlessUsageOverviewResponse, type ServerlessUsageRowsResponse, type ServerlessUsageTimeseriesResponse, type ServerlessUsageDailyResponse, type ServerlessSavingsTotals, type ServerlessSavingsPeriod, type ServerlessUsageSavingsResponse, } from './resources/inference.js';
|
|
78
|
-
export type { BiOSConfig, PaginatedResponse, ApiErrorBody, AvailableGpuAlternative, CapacityMinimumRequirement, InferenceBookingAccepted, Model, ModelSearchParams, ModelSearchResponse, ModelDetailResponse, InferenceModel, InferenceModelListResponse, ModelConfig, Dataset, DatasetListParams, DatasetListResponse, DatasetUploadParams, DatasetPreview, DatasetPreviewParams, DatasetImportHFParams, DatasetRegisterHFParams, DatasetHubSearchParams, DatasetHubPreviewParams, DatasetValidation, DatasetFormatVariant, DatasetFormatSpec, DatasetFormatSpecs, DatasetStorageUsage, TrainingMethod, RLHFAlgorithm, AdapterType, TrainingJobStatus, TrainingStatusFilter, TrainingCreateParams, TrainingListParams, TrainingListResponse, TrainingJob, TrainingMetrics, TrainingResourceMetrics, TrainingDeviceMetrics, MetricPoint, MetricGraphConfig, TrainingCheckpoint, TrainingLogs, TrainingLogEntry, TrainingStopResponse, TrainingResumeResponse, CanonicalTrainingRequest, TrainingPreflightDataset, TrainingPreflightWarning, TrainingPreflightResponse, TrainingCapabilityChoice, TrainingConfigFieldCapability, TrainingCapabilities, GPUChoice, WalletBalance, Transaction, TransactionListResponse, TransactionListParams, GPUInfo, GPUPricingResponse, GPUOptionsParams, TrainingRecommendParams, TrainingAdvisorVerdict, TrainingAdvisorRecommendation, TrainingAdvisorUserShape, TrainingAdvisorJustification, TrainingAdvisorBasis, GPUOption, GPUOptionSuggestion, GPUOptionsResponse, InferenceStatus, InferenceCreateParams, InferenceUpdateParams, InferenceDeployment, InferenceDeploymentSummary, InferenceCreateResponse, InferenceListResponse, InferenceUpdateResponse, InferenceLifecycleResponse, InferenceDeleteResponse, InferenceGPUOptionMarket, InferenceGPUOption, InferenceGPUAlternative, InferenceGPUOptionsParams, InferenceGPUOptionsResponse, InferenceCanonicalRequest, InferencePreflightWarning, InferencePreflightResponse, AdapterCompatibility, AdapterCompatibilityResponse, AdapterCompatibilityParams, SupportedArchitecture, ArchitectureScope, SupportedArchitecturesResponse, ApiKey, ApiKeyScope, ApiKeyIntrospection, Organization, OrgMember, OrgInvite, Workspace, Integration, IntegrationCreateParams, StorageUsage, StorageObject, PromptTokensDetails, ChatCompletionUsage, ServingKind, TrainingRulePausedReason, TrainingTrigger, TrainingRecipe, TrainingRunState, TrainingVerdict, TrainingDecision,
|
|
78
|
+
export type { BiOSConfig, PaginatedResponse, ApiErrorBody, AvailableGpuAlternative, CapacityMinimumRequirement, InferenceBookingAccepted, Model, ModelSearchParams, ModelSearchResponse, ModelDetailResponse, InferenceModel, InferenceModelListResponse, ModelConfig, Dataset, DatasetListParams, DatasetListResponse, DatasetUploadParams, DatasetPreview, DatasetPreviewParams, DatasetImportHFParams, DatasetRegisterHFParams, DatasetHubSearchParams, DatasetHubPreviewParams, DatasetValidation, DatasetFormatVariant, DatasetFormatSpec, DatasetFormatSpecs, DatasetStorageUsage, TrainingMethod, RLHFAlgorithm, AdapterType, TrainingJobStatus, TrainingStatusFilter, TrainingCreateParams, TrainingListParams, TrainingListResponse, TrainingJob, TrainingMetrics, TrainingResourceMetrics, TrainingDeviceMetrics, MetricPoint, MetricGraphConfig, TrainingCheckpoint, TrainingLogs, TrainingLogEntry, TrainingStopResponse, TrainingResumeResponse, CanonicalTrainingRequest, TrainingPreflightDataset, TrainingPreflightWarning, TrainingPreflightResponse, TrainingCapabilityChoice, TrainingConfigFieldCapability, TrainingCapabilities, GPUChoice, WalletBalance, Transaction, TransactionListResponse, TransactionListParams, GPUInfo, GPUPricingResponse, GPUOptionsParams, TrainingRecommendParams, TrainingAdvisorVerdict, TrainingAdvisorRecommendation, TrainingAdvisorUserShape, TrainingAdvisorJustification, TrainingAdvisorBasis, GPUOption, GPUOptionSuggestion, GPUOptionsResponse, InferenceStatus, InferenceCreateParams, InferenceUpdateParams, InferenceDeployment, InferenceDeploymentSummary, InferenceCreateResponse, InferenceListResponse, InferenceUpdateResponse, InferenceLifecycleResponse, InferenceDeleteResponse, InferenceGPUOptionMarket, InferenceGPUOption, InferenceGPUAlternative, InferenceGPUOptionsParams, InferenceGPUOptionsResponse, InferenceCanonicalRequest, InferencePreflightWarning, InferencePreflightResponse, AdapterCompatibility, AdapterCompatibilityResponse, AdapterCompatibilityParams, SupportedArchitecture, ArchitectureScope, SupportedArchitecturesResponse, ApiKey, ApiKeyScope, ApiKeyIntrospection, Organization, OrgMember, OrgInvite, Workspace, Integration, IntegrationCreateParams, StorageUsage, StorageObject, PromptTokensDetails, ChatCompletionUsage, ServingKind, TrainingRulePausedReason, TrainingTrigger, TrainingRecipe, TrainingRunState, TrainingVerdict, TrainingDecision, EvaluationStatus, EvaluationItemStatus, EvaluationWinner, EvaluationItemWinnerFilter, EvaluationWarningCode, TrainingToolCall, TrainingMessage, TrainingRuleServing, TrainingRunPromoteRequest, TrainingRunRejectRequest, TrainingRunRollbackRequest, TrainingRunCancelRequest, TrainingRunListParams, TrainingRun, TrainingRunSummary, TrainingRunEvent, TrainingRunLinks, TrainingRunActions, EvaluationRef, EvaluationDecoding, EvaluationJudgeDimension, EvaluationDimensionScore, EvaluationWarning, EvaluationMargin, Evaluation, EvaluationItemSide, EvaluationItem, EvaluationItemListParams, BenchmarkStatus, BenchmarkSourceKind, BenchmarkRunStatus, BenchmarkScoring, BenchmarkDimensionScore, Benchmark, BenchmarkItem, BenchmarkRun, BenchmarkHistoryPoint, BenchmarkSource, BenchmarkCreateParams, BenchmarkListParams, BenchmarkItemListParams, BenchmarkHistoryParams, BenchmarkRetireParams, InferenceAlias, InferenceAliasRequest, TrainingRuleDeleteResponse, TrainingRunListResponse, TrainingRunResponse, TrainingRunActionResponse, EvaluationItemsResponse, InferenceAliasListResponse, InferenceAliasResponse, InferenceAliasDeleteResponse, BenchmarkListResponse, BenchmarkItemsResponse, BenchmarkHistoryResponse, TrainingRuleRunRefusalCode, TrainingRuleSaveRefusalCode, TrainingRuleMonthlyLimitRefusal, PipelineStatus, PipelineActiveRun, PipelineVersion, PipelineAttempt, PipelineAttemptTrigger, PipelineAttemptOutcome, Pipeline, PipelineListParams, PipelineListResponse, PipelineResponse, PipelineTrainWhen, PipelineSchedule, PipelineWriteRequest, PipelineMutationResponse, PipelineImportParams, PipelineFloors, LoopModel, LoopModelsResponse, LoopTracePipeline, LoopTracePipelinesResponse, LoopMetricsHistoryParams, LoopMetricPoint, LoopSignalDay, LoopMetricsHistory, PromotionPolicyState, RuleBenchmark, PipelineBenchmarksResponse, PipelineBenchmarkAttachParams, PipelineBenchmarkMoveParams, CompareBenchmark, VersionBenchmarkScore, PromotionMeasurement, PromotionOutcome, VersionScores, VersionScoreDifference, VersionCompareResponse, VersionCompareParams, } from './types.js';
|
|
79
79
|
/** The closed set of run states a training run never leaves. */
|
|
80
80
|
export { TERMINAL_RUN_STATES } from './types.js';
|
|
81
81
|
/** The most versions a pipeline may be set to make. */
|
package/dist/index.js
CHANGED
|
@@ -36,7 +36,7 @@ import { Loop } from './resources/loop.js';
|
|
|
36
36
|
* SDK version. Sent as part of the User-Agent header.
|
|
37
37
|
* Must match package.json "version" -- enforced by a contract test.
|
|
38
38
|
*/
|
|
39
|
-
export const VERSION = '0.2.
|
|
39
|
+
export const VERSION = '0.2.20-dev.283';
|
|
40
40
|
export class RunBiOS {
|
|
41
41
|
/** Search models, fetch configs, check adapter compatibility. */
|
|
42
42
|
models;
|
|
@@ -336,7 +336,7 @@ export class Inference {
|
|
|
336
336
|
this.key = config.inferenceKey || envInferenceKey() || envApiKey();
|
|
337
337
|
// Same default host as HttpClient (api.runbios.ai); api.runbios.ai cutover
|
|
338
338
|
// is planned once its DNS exists — update both call sites together.
|
|
339
|
-
this.baseUrl = (config.baseUrl || envBaseUrl() || 'https://api.runbios.ai').replace(/\/+$/, '');
|
|
339
|
+
this.baseUrl = (config.baseUrl || envBaseUrl() || 'https://api-dev.runbios.ai').replace(/\/+$/, '');
|
|
340
340
|
this.timeout = config.timeout ?? 900_000;
|
|
341
341
|
this._http = http;
|
|
342
342
|
}
|
package/dist/resources/loop.d.ts
CHANGED
|
@@ -88,10 +88,10 @@ export declare class Loop {
|
|
|
88
88
|
* a corpus from a pile.
|
|
89
89
|
*
|
|
90
90
|
* Each row is read for what it actually is. A `prompt`/`chosen`/`rejected`
|
|
91
|
-
* triple becomes a
|
|
92
|
-
* thumbs verdict; a question and an answer waits for
|
|
93
|
-
* no answer waits for an answer; a paragraph of prose
|
|
94
|
-
* is not a conversation.
|
|
91
|
+
* triple becomes a correction (`chosen` is learned from); an answer plus a
|
|
92
|
+
* yes-or-no becomes a thumbs verdict; a question and an answer waits for
|
|
93
|
+
* review; a question with no answer waits for an answer; a paragraph of prose
|
|
94
|
+
* is refused, because it is not a conversation.
|
|
95
95
|
*
|
|
96
96
|
* A verdict that arrives *with* the file is kept -- discarding somebody's
|
|
97
97
|
* judgement would be worse -- but it is recorded as having come from your
|
|
@@ -138,7 +138,14 @@ export declare class Loop {
|
|
|
138
138
|
importRows(params: LoopImportParams): Promise<LoopImportResult>;
|
|
139
139
|
/** List captured conversations, newest first. */
|
|
140
140
|
listTraces(params?: LoopTraceListParams): Promise<LoopTraceListResponse>;
|
|
141
|
-
/**
|
|
141
|
+
/**
|
|
142
|
+
* Read one conversation, with every verdict and tag recorded on it.
|
|
143
|
+
*
|
|
144
|
+
* A conversation kept in your own storage carries `payload_stored_externally`.
|
|
145
|
+
* When its text could not be read just now it also carries
|
|
146
|
+
* `payload_unavailable`, with `messages` null and `completion` empty: it is
|
|
147
|
+
* not an empty conversation. Read it again later rather than reviewing it.
|
|
148
|
+
*/
|
|
142
149
|
getTrace(id: string): Promise<LoopTrace>;
|
|
143
150
|
/**
|
|
144
151
|
* Delete a conversation.
|
|
@@ -158,11 +165,11 @@ export declare class Loop {
|
|
|
158
165
|
* Append-only: posting a second verdict does not replace the first. Two
|
|
159
166
|
* reviewers disagreeing about an answer is information worth keeping.
|
|
160
167
|
*
|
|
161
|
-
*
|
|
162
|
-
*
|
|
163
|
-
* `
|
|
164
|
-
*
|
|
165
|
-
*
|
|
168
|
+
* What a pipeline trains on (every pipeline fine-tunes on examples): an
|
|
169
|
+
* `accepted` answer is an example as it stands; the `correction` of an
|
|
170
|
+
* `edited` or `gold` verdict is the example, in place of the model's answer;
|
|
171
|
+
* a `rejected` answer is left out. Nothing is learned from an answer that was
|
|
172
|
+
* rejected or replaced. `source` defaults to `human`.
|
|
166
173
|
*/
|
|
167
174
|
signal(traceId: string, params: LoopSignalParams): Promise<LoopSignal>;
|
|
168
175
|
/**
|
|
@@ -182,23 +189,18 @@ export declare class Loop {
|
|
|
182
189
|
/**
|
|
183
190
|
* The model was wrong; here is what it should have said.
|
|
184
191
|
*
|
|
185
|
-
* The most valuable feedback there is
|
|
186
|
-
*
|
|
187
|
-
* avoid its own.
|
|
192
|
+
* The most valuable feedback there is: your answer is trained on in place of
|
|
193
|
+
* the model's. Nothing is learned from the answer it replaced.
|
|
188
194
|
*/
|
|
189
195
|
correct(traceId: string, answer: string, reason?: string): Promise<LoopSignal>;
|
|
190
196
|
/**
|
|
191
197
|
* Record the GOLD answer for this question — the reference, regardless of
|
|
192
198
|
* what the model happened to say.
|
|
193
199
|
*
|
|
194
|
-
*
|
|
195
|
-
*
|
|
196
|
-
*
|
|
197
|
-
*
|
|
198
|
-
*
|
|
199
|
-
* It is the most reusable thing you can record: it trains SFT, forms a DPO
|
|
200
|
-
* pair when it differs from the answer, and stands in as the GRPO reference
|
|
201
|
-
* when you have not supplied a separate ground truth.
|
|
200
|
+
* A pipeline trains on it exactly as it does on a correction: the gold
|
|
201
|
+
* answer is the example, in place of the model's. The difference is only
|
|
202
|
+
* what it says: a correction says the model was wrong, a gold answer says
|
|
203
|
+
* nothing about the model.
|
|
202
204
|
*/
|
|
203
205
|
gold(traceId: string, answer: string, reason?: string): Promise<LoopSignal>;
|
|
204
206
|
/** Every verdict on one conversation, oldest first. */
|
|
@@ -356,8 +358,8 @@ export declare class Loop {
|
|
|
356
358
|
*
|
|
357
359
|
* `status` says whether it is the platform working (`running`), the member
|
|
358
360
|
* who has to act (`needs_review`, `needs_funds`), or nothing at all until
|
|
359
|
-
* somebody does (`paused`, with `paused_reason`
|
|
360
|
-
*
|
|
361
|
+
* somebody does (`paused`, with `paused_reason` saying why; `complete` at
|
|
362
|
+
* `max_versions`). `month_spent_cents` is the figure the
|
|
361
363
|
* monthly limit is enforced against: what runs started this month have cost
|
|
362
364
|
* at most, not an exact spend.
|
|
363
365
|
*/
|
|
@@ -400,7 +402,9 @@ export declare class Loop {
|
|
|
400
402
|
* numbers `minimums` shows. 503 `AGENT_COULD_NOT_START` means the key the
|
|
401
403
|
* attempt would be scored with could not be renewed just then: nothing was
|
|
402
404
|
* started, so press train now again in a minute. Every refusal is a
|
|
403
|
-
* `TrainingRuleRunRefusalCode`.
|
|
405
|
+
* `TrainingRuleRunRefusalCode`. When the attempt in flight is waiting for
|
|
406
|
+
* GPUs to compare (`active_run.state` `waiting_gpus`), this books its two
|
|
407
|
+
* comparison machines again at once instead of starting another attempt.
|
|
404
408
|
*/
|
|
405
409
|
runPipeline(id: string): Promise<PipelineMutationResponse>;
|
|
406
410
|
/**
|
package/dist/resources/loop.js
CHANGED
|
@@ -91,10 +91,10 @@ export class Loop {
|
|
|
91
91
|
* a corpus from a pile.
|
|
92
92
|
*
|
|
93
93
|
* Each row is read for what it actually is. A `prompt`/`chosen`/`rejected`
|
|
94
|
-
* triple becomes a
|
|
95
|
-
* thumbs verdict; a question and an answer waits for
|
|
96
|
-
* no answer waits for an answer; a paragraph of prose
|
|
97
|
-
* is not a conversation.
|
|
94
|
+
* triple becomes a correction (`chosen` is learned from); an answer plus a
|
|
95
|
+
* yes-or-no becomes a thumbs verdict; a question and an answer waits for
|
|
96
|
+
* review; a question with no answer waits for an answer; a paragraph of prose
|
|
97
|
+
* is refused, because it is not a conversation.
|
|
98
98
|
*
|
|
99
99
|
* A verdict that arrives *with* the file is kept -- discarding somebody's
|
|
100
100
|
* judgement would be worse -- but it is recorded as having come from your
|
|
@@ -182,7 +182,14 @@ export class Loop {
|
|
|
182
182
|
const qs = q.toString();
|
|
183
183
|
return this._http.fetchGet(`/api/loop/traces${qs ? `?${qs}` : ''}`);
|
|
184
184
|
}
|
|
185
|
-
/**
|
|
185
|
+
/**
|
|
186
|
+
* Read one conversation, with every verdict and tag recorded on it.
|
|
187
|
+
*
|
|
188
|
+
* A conversation kept in your own storage carries `payload_stored_externally`.
|
|
189
|
+
* When its text could not be read just now it also carries
|
|
190
|
+
* `payload_unavailable`, with `messages` null and `completion` empty: it is
|
|
191
|
+
* not an empty conversation. Read it again later rather than reviewing it.
|
|
192
|
+
*/
|
|
186
193
|
async getTrace(id) {
|
|
187
194
|
const res = await this._http.fetchGet(`/api/loop/traces/${encodeURIComponent(id)}`);
|
|
188
195
|
return res.trace;
|
|
@@ -205,11 +212,11 @@ export class Loop {
|
|
|
205
212
|
* Append-only: posting a second verdict does not replace the first. Two
|
|
206
213
|
* reviewers disagreeing about an answer is information worth keeping.
|
|
207
214
|
*
|
|
208
|
-
*
|
|
209
|
-
*
|
|
210
|
-
* `
|
|
211
|
-
*
|
|
212
|
-
*
|
|
215
|
+
* What a pipeline trains on (every pipeline fine-tunes on examples): an
|
|
216
|
+
* `accepted` answer is an example as it stands; the `correction` of an
|
|
217
|
+
* `edited` or `gold` verdict is the example, in place of the model's answer;
|
|
218
|
+
* a `rejected` answer is left out. Nothing is learned from an answer that was
|
|
219
|
+
* rejected or replaced. `source` defaults to `human`.
|
|
213
220
|
*/
|
|
214
221
|
async signal(traceId, params) {
|
|
215
222
|
const res = await this._http.fetchPost(`/api/loop/traces/${encodeURIComponent(traceId)}/signals`, params);
|
|
@@ -236,9 +243,8 @@ export class Loop {
|
|
|
236
243
|
/**
|
|
237
244
|
* The model was wrong; here is what it should have said.
|
|
238
245
|
*
|
|
239
|
-
* The most valuable feedback there is
|
|
240
|
-
*
|
|
241
|
-
* avoid its own.
|
|
246
|
+
* The most valuable feedback there is: your answer is trained on in place of
|
|
247
|
+
* the model's. Nothing is learned from the answer it replaced.
|
|
242
248
|
*/
|
|
243
249
|
async correct(traceId, answer, reason) {
|
|
244
250
|
return this.signal(traceId, { verdict: 'edited', correction: answer, reason });
|
|
@@ -247,14 +253,10 @@ export class Loop {
|
|
|
247
253
|
* Record the GOLD answer for this question — the reference, regardless of
|
|
248
254
|
* what the model happened to say.
|
|
249
255
|
*
|
|
250
|
-
*
|
|
251
|
-
*
|
|
252
|
-
*
|
|
253
|
-
*
|
|
254
|
-
*
|
|
255
|
-
* It is the most reusable thing you can record: it trains SFT, forms a DPO
|
|
256
|
-
* pair when it differs from the answer, and stands in as the GRPO reference
|
|
257
|
-
* when you have not supplied a separate ground truth.
|
|
256
|
+
* A pipeline trains on it exactly as it does on a correction: the gold
|
|
257
|
+
* answer is the example, in place of the model's. The difference is only
|
|
258
|
+
* what it says: a correction says the model was wrong, a gold answer says
|
|
259
|
+
* nothing about the model.
|
|
258
260
|
*/
|
|
259
261
|
async gold(traceId, answer, reason) {
|
|
260
262
|
return this.signal(traceId, { verdict: 'gold', correction: answer, reason });
|
|
@@ -491,8 +493,8 @@ export class Loop {
|
|
|
491
493
|
*
|
|
492
494
|
* `status` says whether it is the platform working (`running`), the member
|
|
493
495
|
* who has to act (`needs_review`, `needs_funds`), or nothing at all until
|
|
494
|
-
* somebody does (`paused`, with `paused_reason`
|
|
495
|
-
*
|
|
496
|
+
* somebody does (`paused`, with `paused_reason` saying why; `complete` at
|
|
497
|
+
* `max_versions`). `month_spent_cents` is the figure the
|
|
496
498
|
* monthly limit is enforced against: what runs started this month have cost
|
|
497
499
|
* at most, not an exact spend.
|
|
498
500
|
*/
|
|
@@ -553,7 +555,9 @@ export class Loop {
|
|
|
553
555
|
* numbers `minimums` shows. 503 `AGENT_COULD_NOT_START` means the key the
|
|
554
556
|
* attempt would be scored with could not be renewed just then: nothing was
|
|
555
557
|
* started, so press train now again in a minute. Every refusal is a
|
|
556
|
-
* `TrainingRuleRunRefusalCode`.
|
|
558
|
+
* `TrainingRuleRunRefusalCode`. When the attempt in flight is waiting for
|
|
559
|
+
* GPUs to compare (`active_run.state` `waiting_gpus`), this books its two
|
|
560
|
+
* comparison machines again at once instead of starting another attempt.
|
|
557
561
|
*/
|
|
558
562
|
async runPipeline(id) {
|
|
559
563
|
return this._http.fetchPost(`/api/loop/pipelines/${encodeURIComponent(id)}/run`, {});
|
package/dist/types.d.ts
CHANGED
|
@@ -8,13 +8,13 @@ export interface BiOSConfig {
|
|
|
8
8
|
orgId?: string;
|
|
9
9
|
/** Workspace ID. Optional when using API keys (resolved from the key). Can override for multi-workspace keys. */
|
|
10
10
|
workspaceId?: string;
|
|
11
|
-
/** Base URL for the API. Falls back to the RUNBIOS_BASE_URL environment variable (legacy: BIOS_BASE_URL), then the canonical https://api.runbios.ai hostname. */
|
|
11
|
+
/** Base URL for the API. Falls back to the RUNBIOS_BASE_URL environment variable (legacy: BIOS_BASE_URL), then the canonical https://api-dev.runbios.ai hostname. */
|
|
12
12
|
baseUrl?: string;
|
|
13
13
|
/** Request timeout in milliseconds. Defaults to 30000. */
|
|
14
14
|
timeout?: number;
|
|
15
15
|
/** Default per-deployment inference key. Can be overridden per inference call. */
|
|
16
16
|
inferenceKey?: string;
|
|
17
|
-
/** Inference base URL. Defaults to baseUrl, then https://api.runbios.ai. */
|
|
17
|
+
/** Inference base URL. Defaults to baseUrl, then https://api-dev.runbios.ai. */
|
|
18
18
|
inferenceBaseUrl?: string;
|
|
19
19
|
/** End-to-end inference timeout in milliseconds. Defaults to 15 minutes. */
|
|
20
20
|
inferenceTimeout?: number;
|
|
@@ -781,10 +781,24 @@ export interface TrainingJob {
|
|
|
781
781
|
checkpoint_count?: number;
|
|
782
782
|
/** The current machine attempt's per-phase records, in canonical order. */
|
|
783
783
|
phases?: TrainingJobPhase[];
|
|
784
|
-
/**
|
|
784
|
+
/**
|
|
785
|
+
* Estimated seconds remaining (absent when unknown). What it covers is
|
|
786
|
+
* `eta_scope`: before step 1 only the time until training starts.
|
|
787
|
+
*/
|
|
785
788
|
eta_seconds?: number;
|
|
786
|
-
/**
|
|
789
|
+
/**
|
|
790
|
+
* How the ETA was derived: `measured` from this job's own live rate
|
|
791
|
+
* (training: its own steps since the first), or `historical` from past runs
|
|
792
|
+
* of the same model on the same GPU (else the GPU).
|
|
793
|
+
*/
|
|
787
794
|
eta_confidence?: 'measured' | 'historical';
|
|
795
|
+
/**
|
|
796
|
+
* What `eta_seconds` covers: `until_training` before step 1 (startup
|
|
797
|
+
* phases only; training time is measured from the run's own steps, never
|
|
798
|
+
* guessed), `run` once training has started (remaining training plus what
|
|
799
|
+
* follows it).
|
|
800
|
+
*/
|
|
801
|
+
eta_scope?: 'until_training' | 'run';
|
|
788
802
|
/** Most recent per-phase update time. */
|
|
789
803
|
last_phase_update_at?: string;
|
|
790
804
|
/**
|
|
@@ -1663,6 +1677,12 @@ export interface InferenceDeployment {
|
|
|
1663
1677
|
base_model_revision?: string;
|
|
1664
1678
|
immutable_base_model_source?: string;
|
|
1665
1679
|
serving_mode?: 'full' | 'adapter' | 'merged';
|
|
1680
|
+
/**
|
|
1681
|
+
* Only on an adapter deployment: the `model` that asks its BASE weights with
|
|
1682
|
+
* no adapter, `<name>:base`, on /chat/completions and /completions. A
|
|
1683
|
+
* deployment with no adapter has no separate base and refuses `:base`.
|
|
1684
|
+
*/
|
|
1685
|
+
base_weights_model?: string;
|
|
1666
1686
|
supports_tool_calls?: boolean;
|
|
1667
1687
|
tool_call_parser?: InferenceToolCallParser | null;
|
|
1668
1688
|
reasoning_parser?: InferenceReasoningParser | null;
|
|
@@ -2297,18 +2317,17 @@ export interface ChatCompletionUsage {
|
|
|
2297
2317
|
total_tokens: number;
|
|
2298
2318
|
prompt_tokens_details?: PromptTokensDetails;
|
|
2299
2319
|
}
|
|
2300
|
-
/**
|
|
2301
|
-
export type LoopMethod = 'sft' | 'dpo' | 'grpo' | 'kto';
|
|
2302
|
-
/** Who judged an answer. */
|
|
2320
|
+
/** Who gave a verdict. Feedback you send is `human` unless you say otherwise. */
|
|
2303
2321
|
export type LoopSource = 'human' | 'verifier' | 'judge' | 'behavioural';
|
|
2304
|
-
/** The judgement itself. */
|
|
2305
2322
|
/**
|
|
2306
2323
|
* The judgement recorded on an answer.
|
|
2307
2324
|
*
|
|
2308
|
-
* `accepted` / `rejected` are the thumbs (see `upvote` / `downvote`)
|
|
2309
|
-
*
|
|
2325
|
+
* `accepted` / `rejected` are the thumbs (see `upvote` / `downvote`): an
|
|
2326
|
+
* accepted answer is trained on, a rejected one is left out.
|
|
2327
|
+
* `edited` says the model was WRONG and carries the better answer, which is
|
|
2328
|
+
* trained on in place of the model's.
|
|
2310
2329
|
* `gold` records the reference answer for the question, making no claim about
|
|
2311
|
-
* whether the model was right
|
|
2330
|
+
* whether the model was right; it is trained on the same way.
|
|
2312
2331
|
*/
|
|
2313
2332
|
export type LoopVerdict = 'accepted' | 'rejected' | 'edited' | 'scored' | 'gold';
|
|
2314
2333
|
/** A function an assistant turn asked to invoke. `arguments` is JSON-encoded. */
|
|
@@ -2495,7 +2514,7 @@ export interface LoopImportResult {
|
|
|
2495
2514
|
not_saved_rows?: number[];
|
|
2496
2515
|
/**
|
|
2497
2516
|
* Rows whose CONVERSATION stored and whose VERDICT did not, and which rows
|
|
2498
|
-
* those were. A
|
|
2517
|
+
* those were. A chosen/rejected row or a thumbs label is two writes, and the
|
|
2499
2518
|
* second can fail on its own: the conversation is then in the loop carrying
|
|
2500
2519
|
* nobody's judgement, counted under `needs_review` rather than `reviewed`,
|
|
2501
2520
|
* and not trainable. The call rejects with {@link LoopImportIncompleteError},
|
|
@@ -2532,6 +2551,8 @@ export interface LoopSignal {
|
|
|
2532
2551
|
ground_truth: string | null;
|
|
2533
2552
|
reason: string | null;
|
|
2534
2553
|
author: string | null;
|
|
2554
|
+
/** What the caller attached when recording it, or null. */
|
|
2555
|
+
metadata: Record<string, unknown> | null;
|
|
2535
2556
|
created_at: string;
|
|
2536
2557
|
}
|
|
2537
2558
|
export interface LoopSignalParams {
|
|
@@ -2539,9 +2560,9 @@ export interface LoopSignalParams {
|
|
|
2539
2560
|
source?: LoopSource;
|
|
2540
2561
|
/** Required when verdict is `edited`. The answer the model should have given. */
|
|
2541
2562
|
correction?: string;
|
|
2542
|
-
/** Required when verdict is `scored
|
|
2563
|
+
/** Required when verdict is `scored`: above zero trains the answer as it stands, otherwise it is left out. */
|
|
2543
2564
|
score?: number;
|
|
2544
|
-
/** A value or fact the answer can be checked against.
|
|
2565
|
+
/** A value or fact the answer can be checked against. Kept with the verdict; a pipeline trains on `correction`, not on this. */
|
|
2545
2566
|
ground_truth?: string;
|
|
2546
2567
|
reason?: string;
|
|
2547
2568
|
author?: string;
|
|
@@ -2579,7 +2600,11 @@ export interface LoopTrace {
|
|
|
2579
2600
|
deployment_id: string | null;
|
|
2580
2601
|
model: string;
|
|
2581
2602
|
model_version: string | null;
|
|
2582
|
-
|
|
2603
|
+
/**
|
|
2604
|
+
* The conversation. NULL, with `completion` empty, when its text is kept in
|
|
2605
|
+
* your own storage and could not be read just now: see `payload_unavailable`.
|
|
2606
|
+
*/
|
|
2607
|
+
messages: LoopMessage[] | null;
|
|
2583
2608
|
completion: string;
|
|
2584
2609
|
tool_calls?: LoopToolCall[];
|
|
2585
2610
|
tools?: unknown;
|
|
@@ -2591,9 +2616,40 @@ export interface LoopTrace {
|
|
|
2591
2616
|
redacted_at: string | null;
|
|
2592
2617
|
/** What the redaction pass removed, counted by rule. */
|
|
2593
2618
|
redaction_report?: Record<string, number>;
|
|
2619
|
+
/** What the caller attached when recording it, or null. */
|
|
2620
|
+
metadata: Record<string, unknown> | null;
|
|
2594
2621
|
created_at: string;
|
|
2595
2622
|
expires_at: string;
|
|
2623
|
+
/** Every verdict recorded on it, oldest first. On the single read only. */
|
|
2596
2624
|
signals?: LoopSignal[];
|
|
2625
|
+
/** How many verdicts it carries. */
|
|
2626
|
+
signal_count: number;
|
|
2627
|
+
/**
|
|
2628
|
+
* The verdict that decides it -- the one training uses, not whichever
|
|
2629
|
+
* arrived last. Absent when nobody has given feedback.
|
|
2630
|
+
*/
|
|
2631
|
+
verdict?: LoopVerdict;
|
|
2632
|
+
/** Who gave that verdict. */
|
|
2633
|
+
verdict_source?: LoopSource;
|
|
2634
|
+
/**
|
|
2635
|
+
* The score a `scored` verdict carries: above zero it trains the answer as
|
|
2636
|
+
* it stands (good feedback), zero or below it trains nothing (bad).
|
|
2637
|
+
*/
|
|
2638
|
+
verdict_score?: number;
|
|
2639
|
+
/** Its tags. On the single read only. */
|
|
2640
|
+
labels?: LoopLabel[];
|
|
2641
|
+
/**
|
|
2642
|
+
* True when the conversation's text is kept in your own storage rather than
|
|
2643
|
+
* here. The text is read from there on every request.
|
|
2644
|
+
*/
|
|
2645
|
+
payload_stored_externally?: boolean;
|
|
2646
|
+
/**
|
|
2647
|
+
* True when that text could not be read just now (the storage could not be
|
|
2648
|
+
* reached, or the object is gone). `messages` is then null and `completion`
|
|
2649
|
+
* empty: the conversation is NOT empty. Try again later; do not review it or
|
|
2650
|
+
* treat it as a blank answer.
|
|
2651
|
+
*/
|
|
2652
|
+
payload_unavailable?: boolean;
|
|
2597
2653
|
}
|
|
2598
2654
|
export interface LoopTraceListParams {
|
|
2599
2655
|
deployment_id?: string;
|
|
@@ -2810,11 +2866,13 @@ export type TrainingRulePausedReason = 'consent_invalid' | 'insufficient_funds'
|
|
|
2810
2866
|
*/
|
|
2811
2867
|
| 'training_not_enabled';
|
|
2812
2868
|
/**
|
|
2813
|
-
* Why a run fired
|
|
2814
|
-
*
|
|
2815
|
-
*
|
|
2869
|
+
* Why a run fired, as the run stores it: `rows` (enough new samples),
|
|
2870
|
+
* `cadence` (the schedule), `both`, `manual` (train now), `challenger` (the
|
|
2871
|
+
* pipeline's challenger model trained on the same data as the attempt before
|
|
2872
|
+
* it), or `variant` (a recipe variant, retired: the same base model on the
|
|
2873
|
+
* same conversations with one training setting changed; old runs only).
|
|
2816
2874
|
*/
|
|
2817
|
-
export type TrainingTrigger = 'rows' | 'cadence' | 'both' | 'manual' | 'variant';
|
|
2875
|
+
export type TrainingTrigger = 'rows' | 'cadence' | 'both' | 'manual' | 'variant' | 'challenger';
|
|
2818
2876
|
/**
|
|
2819
2877
|
* The four training settings a recipe variant may change, as the run was
|
|
2820
2878
|
* actually trained: the training service's own defaults are filled in where
|
|
@@ -2828,12 +2886,22 @@ export interface TrainingRecipe {
|
|
|
2828
2886
|
lora_rank: number | null;
|
|
2829
2887
|
lora_alpha: number | null;
|
|
2830
2888
|
}
|
|
2831
|
-
export type TrainingRunState = 'built' | 'registering' | 'dataset_validating' | 'preflighting' | 'job_creating' | 'training' | 'checkpoint_ready' | 'candidate_booking' | 'candidate_running' | 'evaluating' | 'reported' | 'awaiting_review' | 'promoting' | 'waiting_funds'
|
|
2889
|
+
export type TrainingRunState = 'built' | 'registering' | 'dataset_validating' | 'preflighting' | 'job_creating' | 'training' | 'checkpoint_ready' | 'candidate_booking' | 'candidate_running' | 'evaluating' | 'reported' | 'awaiting_review' | 'promoting' | 'waiting_funds'
|
|
2890
|
+
/**
|
|
2891
|
+
* Trained, and waiting for GPUs to run the new model and the one it is
|
|
2892
|
+
* compared with side by side: one machine got a GPU and the other did not
|
|
2893
|
+
* within 20 minutes, so both were stopped. Each try can bill the machine
|
|
2894
|
+
* that got a GPU for up to those 20 minutes; nothing bills between tries.
|
|
2895
|
+
* It keeps what it trained, books both again after 30 minutes, then 1, 2
|
|
2896
|
+
* and 4 hours, and ends (`failed`, `COMPARISON_NO_GPUS`) after four tries or
|
|
2897
|
+
* 24 hours (`COMPARISON_WORKSPACE_FULL` when what was missing was room in
|
|
2898
|
+
* the workspace). Train now books them again at once.
|
|
2899
|
+
*/
|
|
2900
|
+
| 'waiting_gpus' | 'promoted' | 'rejected' | 'expired' | 'failed' | 'budget_stopped' | 'cancelled' | 'superseded' | 'rolled_back';
|
|
2832
2901
|
/** The closed set a run never leaves. */
|
|
2833
2902
|
export declare const TERMINAL_RUN_STATES: readonly TrainingRunState[];
|
|
2834
2903
|
export type TrainingVerdict = 'better' | 'not_better' | 'inconclusive' | 'not_evaluated';
|
|
2835
2904
|
export type TrainingDecision = 'auto_promoted' | 'promoted' | 'rejected' | 'auto_rejected' | 'rolled_back';
|
|
2836
|
-
export type NotifyState = 'none' | 'pending' | 'sending' | 'sent' | 'suppressed' | 'failed';
|
|
2837
2905
|
export type EvaluationStatus = 'open' | 'done' | 'failed' | 'budget_stopped';
|
|
2838
2906
|
export type EvaluationItemStatus = 'pending' | 'working' | 'done' | 'failed';
|
|
2839
2907
|
export type EvaluationWinner = 'candidate' | 'incumbent' | 'tie';
|
|
@@ -2891,8 +2959,6 @@ export interface TrainingRun {
|
|
|
2891
2959
|
workspace_id: string;
|
|
2892
2960
|
seq: number;
|
|
2893
2961
|
rule_revision: number;
|
|
2894
|
-
authorized_by: string;
|
|
2895
|
-
rule_snapshot: Record<string, unknown>;
|
|
2896
2962
|
trigger: TrainingTrigger;
|
|
2897
2963
|
/**
|
|
2898
2964
|
* For a recipe variant, what it changed, in plain words ("half the learning
|
|
@@ -2921,13 +2987,21 @@ export interface TrainingRun {
|
|
|
2921
2987
|
* no remedy worth printing.
|
|
2922
2988
|
*/
|
|
2923
2989
|
error_next_step: string | null;
|
|
2924
|
-
training_ceiling_cents: number;
|
|
2925
|
-
candidate_ceiling_cents: number;
|
|
2926
|
-
eval_ceiling_cents: number;
|
|
2927
2990
|
billed_training_cents: number;
|
|
2991
|
+
/**
|
|
2992
|
+
* The comparison machines: each from the first time it was on a GPU to its
|
|
2993
|
+
* stop, less any time back in the GPU queue, at the price the platform
|
|
2994
|
+
* placed it at.
|
|
2995
|
+
*/
|
|
2928
2996
|
billed_candidate_cents: number;
|
|
2997
|
+
/**
|
|
2998
|
+
* True when a comparison machine was charged: the figure is then the most
|
|
2999
|
+
* it can have cost ("up to"), since it counts each machine from its first
|
|
3000
|
+
* boot where the platform bills from its first answer (and at the hourly
|
|
3001
|
+
* cap when no placement price was reported).
|
|
3002
|
+
*/
|
|
3003
|
+
billed_candidate_up_to: boolean;
|
|
2929
3004
|
spent_eval_cents: number;
|
|
2930
|
-
loop_dataset_id: string | null;
|
|
2931
3005
|
train_rows: number | null;
|
|
2932
3006
|
holdout_rows: number | null;
|
|
2933
3007
|
train_dataset_id: string | null;
|
|
@@ -2967,8 +3041,7 @@ export interface TrainingRun {
|
|
|
2967
3041
|
benchmark_run_id: string | null;
|
|
2968
3042
|
/**
|
|
2969
3043
|
* What the replay's calls cost. Kept apart from `spent_eval_cents` so a
|
|
2970
|
-
* report can say what the comparison cost and what the benchmark cost
|
|
2971
|
-
* come out of `eval_ceiling_cents` and their sum can never exceed it, so
|
|
3044
|
+
* report can say what the comparison cost and what the benchmark cost, so
|
|
2972
3045
|
* anything totalling what a run cost has to add this one too.
|
|
2973
3046
|
*/
|
|
2974
3047
|
benchmark_spent_cents: number;
|
|
@@ -2982,10 +3055,6 @@ export interface TrainingRun {
|
|
|
2982
3055
|
decided_at: string | null;
|
|
2983
3056
|
auto_promote_at: string | null;
|
|
2984
3057
|
review_deadline_at: string | null;
|
|
2985
|
-
notify_state: NotifyState;
|
|
2986
|
-
notify_event: string | null;
|
|
2987
|
-
notify_error: string | null;
|
|
2988
|
-
notify_attempts: number;
|
|
2989
3058
|
created_at: string;
|
|
2990
3059
|
updated_at: string;
|
|
2991
3060
|
finished_at: string | null;
|
|
@@ -3023,6 +3092,8 @@ export interface TrainingRunSummary {
|
|
|
3023
3092
|
holdout_rows: number | null;
|
|
3024
3093
|
billed_training_cents: number;
|
|
3025
3094
|
billed_candidate_cents: number;
|
|
3095
|
+
/** See {@link TrainingRun.billed_candidate_up_to}. */
|
|
3096
|
+
billed_candidate_up_to: boolean;
|
|
3026
3097
|
spent_eval_cents: number;
|
|
3027
3098
|
/** The fourth money column. A row that leaves it out adds up short. */
|
|
3028
3099
|
benchmark_spent_cents: number;
|
|
@@ -3031,7 +3102,6 @@ export interface TrainingRunSummary {
|
|
|
3031
3102
|
evaluation_id: string | null;
|
|
3032
3103
|
review_deadline_at: string | null;
|
|
3033
3104
|
auto_promote_at: string | null;
|
|
3034
|
-
notify_state: NotifyState;
|
|
3035
3105
|
created_at: string;
|
|
3036
3106
|
finished_at: string | null;
|
|
3037
3107
|
}
|
|
@@ -3049,7 +3119,6 @@ export interface TrainingRunEvent {
|
|
|
3049
3119
|
export interface TrainingRunLinks {
|
|
3050
3120
|
training_job_url: string | null;
|
|
3051
3121
|
candidate_url: string | null;
|
|
3052
|
-
dataset_url: string | null;
|
|
3053
3122
|
evaluation_url: string | null;
|
|
3054
3123
|
/** The TREND the benchmark number belongs to, not the one replay. */
|
|
3055
3124
|
benchmark_url: string | null;
|
|
@@ -3091,41 +3160,13 @@ export interface EvaluationWarning {
|
|
|
3091
3160
|
export interface EvaluationMargin {
|
|
3092
3161
|
promote_margin: number;
|
|
3093
3162
|
promote_min_win_rate: number;
|
|
3094
|
-
min_judge_agreement: number;
|
|
3095
3163
|
min_holdout_rows: number;
|
|
3096
3164
|
}
|
|
3097
|
-
/**
|
|
3098
|
-
* One rubric dimension of a judge-versus-human agreement.
|
|
3099
|
-
*
|
|
3100
|
-
* Deliberately not EvaluationDimensionScore: that type carries `incumbent` and
|
|
3101
|
-
* `candidate`, which are the two models being compared, and an agreement has
|
|
3102
|
-
* neither. `pairs` is per dimension because a reviewer who rated one dimension
|
|
3103
|
-
* and skipped another leaves a different denominator behind each number.
|
|
3104
|
-
*/
|
|
3105
|
-
export interface JudgeAgreementDimension {
|
|
3106
|
-
dimension: string;
|
|
3107
|
-
pairs: number;
|
|
3108
|
-
agreement: number | null;
|
|
3109
|
-
mean_abs_error: number | null;
|
|
3110
|
-
}
|
|
3111
|
-
/** How often this judge agreed with the workspace's own reviewers. */
|
|
3112
|
-
export interface JudgeAgreement {
|
|
3113
|
-
judge_id: string;
|
|
3114
|
-
pairs: number;
|
|
3115
|
-
agreement: number | null;
|
|
3116
|
-
mean_abs_error: number | null;
|
|
3117
|
-
per_dimension: JudgeAgreementDimension[];
|
|
3118
|
-
window_days: number;
|
|
3119
|
-
computed_at: string;
|
|
3120
|
-
/** "Not enough reviewer overlap yet" is an answer; 100% of two is not. */
|
|
3121
|
-
enough_pairs: boolean;
|
|
3122
|
-
}
|
|
3123
3165
|
/** The comparison report: the candidate against what serves today, same rows. */
|
|
3124
3166
|
export interface Evaluation {
|
|
3125
3167
|
id: string;
|
|
3126
3168
|
run_id: string;
|
|
3127
3169
|
workspace_id: string;
|
|
3128
|
-
loop_dataset_id: string | null;
|
|
3129
3170
|
split: string;
|
|
3130
3171
|
incumbent_ref: EvaluationRef;
|
|
3131
3172
|
candidate_ref: EvaluationRef;
|
|
@@ -3134,7 +3175,6 @@ export interface Evaluation {
|
|
|
3134
3175
|
judge_model: string | null;
|
|
3135
3176
|
judge_instructions: string | null;
|
|
3136
3177
|
judge_dimensions: EvaluationJudgeDimension[];
|
|
3137
|
-
judge_system_prompt: string | null;
|
|
3138
3178
|
decoding: EvaluationDecoding;
|
|
3139
3179
|
rows_selected: number;
|
|
3140
3180
|
rows_scored: number;
|
|
@@ -3150,7 +3190,6 @@ export interface Evaluation {
|
|
|
3150
3190
|
judge_incumbent_mean: number | null;
|
|
3151
3191
|
judge_candidate_mean: number | null;
|
|
3152
3192
|
per_dimension: EvaluationDimensionScore[];
|
|
3153
|
-
judge_agreement: JudgeAgreement | null;
|
|
3154
3193
|
/** Informational: there is no incumbent counterpart to compare it against. */
|
|
3155
3194
|
trainer_eval_loss: number | null;
|
|
3156
3195
|
warnings: EvaluationWarning[];
|
|
@@ -3283,14 +3322,11 @@ export interface Benchmark {
|
|
|
3283
3322
|
judge_model: string;
|
|
3284
3323
|
judge_instructions: string;
|
|
3285
3324
|
judge_dimensions: EvaluationJudgeDimension[];
|
|
3286
|
-
judge_system_prompt: string | null;
|
|
3287
3325
|
decoding: EvaluationDecoding;
|
|
3288
3326
|
scoring: BenchmarkScoring;
|
|
3289
3327
|
item_count: number;
|
|
3290
3328
|
/** Recomputed from the rows and compared at the start of every replay. */
|
|
3291
3329
|
items_digest: string;
|
|
3292
|
-
/** A cap INSIDE the run's own amount for judge calls, never an addition to it. */
|
|
3293
|
-
per_run_ceiling_cents: number;
|
|
3294
3330
|
created_by: string;
|
|
3295
3331
|
created_at: string;
|
|
3296
3332
|
retired_at: string | null;
|
|
@@ -3342,7 +3378,6 @@ export interface BenchmarkRun {
|
|
|
3342
3378
|
prompt_tokens: number;
|
|
3343
3379
|
completion_tokens: number;
|
|
3344
3380
|
spent_cents: number;
|
|
3345
|
-
ceiling_cents: number;
|
|
3346
3381
|
status: BenchmarkRunStatus;
|
|
3347
3382
|
/** The whole explanation when there is no score, so never empty on one. */
|
|
3348
3383
|
status_reason: string | null;
|
|
@@ -3448,6 +3483,10 @@ export interface BenchmarkRetireParams {
|
|
|
3448
3483
|
* monthly cap, so what this attempt trained could not be scored. The body
|
|
3449
3484
|
* carries `resumes_at`, when the cap lifts (the next UTC month, or sooner
|
|
3450
3485
|
* if the key's cap is raised).
|
|
3486
|
+
* - `LIVE_VERSION_STOPPED` (409): the deployment the pipeline's live version
|
|
3487
|
+
* answers from is stopped, paused until funds are added, deleted or failed,
|
|
3488
|
+
* so an attempt could not be compared against it. Resume it in Deployments,
|
|
3489
|
+
* or roll the version back, then train again.
|
|
3451
3490
|
* - `AGENT_COULD_NOT_START` (503): the key every comparison is scored with is
|
|
3452
3491
|
* missing or was refused, and train now could not renew it just then.
|
|
3453
3492
|
* Nothing was started. Press train now again in a minute.
|
|
@@ -3460,7 +3499,7 @@ export interface BenchmarkRetireParams {
|
|
|
3460
3499
|
* `MONTHLY_LIMIT_REACHED` and `SCORING_CAPPED` once their date has passed;
|
|
3461
3500
|
* retrying any other one unchanged gets the same answer.
|
|
3462
3501
|
*/
|
|
3463
|
-
export type TrainingRuleRunRefusalCode = 'NOT_ENOUGH_SAMPLES' | 'PIPELINE_NEEDS_SAVE' | 'RULE_PAUSED' | 'RUN_ACTIVE' | 'VERSION_LIMIT_REACHED' | 'MONTHLY_LIMIT_REACHED' | 'SCORING_CAPPED' | 'AGENT_COULD_NOT_START' | 'AGENT_OFF';
|
|
3502
|
+
export type TrainingRuleRunRefusalCode = 'NOT_ENOUGH_SAMPLES' | 'PIPELINE_NEEDS_SAVE' | 'RULE_PAUSED' | 'RUN_ACTIVE' | 'VERSION_LIMIT_REACHED' | 'MONTHLY_LIMIT_REACHED' | 'SCORING_CAPPED' | 'LIVE_VERSION_STOPPED' | 'AGENT_COULD_NOT_START' | 'AGENT_OFF';
|
|
3464
3503
|
/**
|
|
3465
3504
|
* The machine codes saving a pipeline (`createPipeline` or `updatePipeline`)
|
|
3466
3505
|
* can be refused with: every one the two routes answer beyond a malformed
|
|
@@ -3520,8 +3559,8 @@ export type TrainingRuleSaveRefusalCode = 'PIPELINE_NAME_TAKEN' | 'OWN_TAG_TAKEN
|
|
|
3520
3559
|
*
|
|
3521
3560
|
* The test is the worst case, not the average: a run may start only if the
|
|
3522
3561
|
* most the rule's runs this UTC calendar month can have cost
|
|
3523
|
-
* (`month_spent_cents`), plus the most the run about to start may spend
|
|
3524
|
-
*
|
|
3562
|
+
* (`month_spent_cents`), plus the most the run about to start may spend
|
|
3563
|
+
* (`run_max_cents`), fits under the limit.
|
|
3525
3564
|
*/
|
|
3526
3565
|
export interface TrainingRuleMonthlyLimitRefusal {
|
|
3527
3566
|
error: {
|
|
@@ -3532,12 +3571,18 @@ export interface TrainingRuleMonthlyLimitRefusal {
|
|
|
3532
3571
|
month_spent_cents: number;
|
|
3533
3572
|
/** The rule's monthly limit. */
|
|
3534
3573
|
monthly_ceiling_cents: number;
|
|
3535
|
-
/**
|
|
3574
|
+
/**
|
|
3575
|
+
* The most one attempt of this rule may spend: its training, its comparison
|
|
3576
|
+
* machine and its judging. While no version is live that machine also
|
|
3577
|
+
* answers as the base model, so a pipeline's own attempt counts one; a
|
|
3578
|
+
* challenger's attempt also counts the base model's own machine beside it.
|
|
3579
|
+
*/
|
|
3536
3580
|
run_max_cents: number;
|
|
3537
3581
|
/**
|
|
3538
3582
|
* The first instant of the next UTC month, when the counted spend starts
|
|
3539
3583
|
* again from nothing. When `run_max_cents` alone exceeds the limit, no month
|
|
3540
|
-
* will ever fit
|
|
3584
|
+
* will ever fit: saving the pipeline sets the limit again from its model's
|
|
3585
|
+
* prices, and nothing else changes it; `message` says which.
|
|
3541
3586
|
*/
|
|
3542
3587
|
resumes_at: string;
|
|
3543
3588
|
}
|
|
@@ -3549,8 +3594,9 @@ export declare const PIPELINE_MAX_VERSIONS_CEILING = 100;
|
|
|
3549
3594
|
* platform working on it.
|
|
3550
3595
|
*
|
|
3551
3596
|
* `paused` is a rule that is switched off, AND an enabled rule the platform
|
|
3552
|
-
* has stopped firing (`paused_reason`
|
|
3553
|
-
*
|
|
3597
|
+
* has stopped firing (`paused_reason`; `consent_invalid` when its settings
|
|
3598
|
+
* changed since it was last saved). `complete` is a pipeline at its
|
|
3599
|
+
* `max_versions`.
|
|
3554
3600
|
*/
|
|
3555
3601
|
export type PipelineStatus = 'running' | 'needs_review' | 'needs_funds' | 'waiting' | 'paused' | 'complete';
|
|
3556
3602
|
/**
|
|
@@ -3574,6 +3620,11 @@ export interface PipelineActiveRun {
|
|
|
3574
3620
|
gpu_seconds: number | null;
|
|
3575
3621
|
/** What it has been billed so far, in cents. */
|
|
3576
3622
|
cost_cents: number;
|
|
3623
|
+
/**
|
|
3624
|
+
* True when a comparison machine was charged in `cost_cents`: the figure is
|
|
3625
|
+
* the most it can have cost, so show it as "up to".
|
|
3626
|
+
*/
|
|
3627
|
+
cost_up_to: boolean;
|
|
3577
3628
|
/** The version number it will take if it is put live. */
|
|
3578
3629
|
will_be_version: number;
|
|
3579
3630
|
/** Always null: nothing in flight is a version yet. */
|
|
@@ -3588,8 +3639,6 @@ export interface PipelineActiveRun {
|
|
|
3588
3639
|
* not a version, and becomes one only if a member promotes it.
|
|
3589
3640
|
*/
|
|
3590
3641
|
competed: boolean;
|
|
3591
|
-
/** What recipe it is trying, when it is a recipe variant. See TrainingRun.recipe_note. */
|
|
3592
|
-
recipe_note: string | null;
|
|
3593
3642
|
}
|
|
3594
3643
|
/**
|
|
3595
3644
|
* One version: an attempt that was put live -- promoted by the pipeline's
|
|
@@ -3648,6 +3697,14 @@ export interface PipelineVersion {
|
|
|
3648
3697
|
* not what it did.
|
|
3649
3698
|
*/
|
|
3650
3699
|
cost_includes_candidate_ceiling: boolean;
|
|
3700
|
+
/**
|
|
3701
|
+
* True when `cost_cents` is the most the version can have cost rather than
|
|
3702
|
+
* what it did: its comparison machines at their ceilings
|
|
3703
|
+
* (`cost_includes_candidate_ceiling`), or one charged (counted from its
|
|
3704
|
+
* first boot, where the platform bills from its first answer). Show it as
|
|
3705
|
+
* "up to".
|
|
3706
|
+
*/
|
|
3707
|
+
cost_up_to: boolean;
|
|
3651
3708
|
/** Machine time the attempt held (training and comparison); null when unknown. */
|
|
3652
3709
|
gpu_seconds: number | null;
|
|
3653
3710
|
created_at: string;
|
|
@@ -3659,18 +3716,10 @@ export interface PipelineVersion {
|
|
|
3659
3716
|
*/
|
|
3660
3717
|
compared_at: string | null;
|
|
3661
3718
|
/**
|
|
3662
|
-
* Why the
|
|
3663
|
-
*
|
|
3664
|
-
* value is a version that learned from newly reviewed conversations (or a
|
|
3665
|
-
* member's run now).
|
|
3719
|
+
* Why the attempt that made it was made, in the words `attempts[]` uses for
|
|
3720
|
+
* the same run; see {@link PipelineAttempt.trigger}.
|
|
3666
3721
|
*/
|
|
3667
|
-
trigger:
|
|
3668
|
-
/** See TrainingRun.recipe_note. */
|
|
3669
|
-
recipe_note: string | null;
|
|
3670
|
-
/** See TrainingRun.recipe_of_version. */
|
|
3671
|
-
recipe_of_version: number | null;
|
|
3672
|
-
/** See TrainingRun.recipe. */
|
|
3673
|
-
recipe: TrainingRecipe;
|
|
3722
|
+
trigger: PipelineAttemptTrigger;
|
|
3674
3723
|
finished_at: string | null;
|
|
3675
3724
|
}
|
|
3676
3725
|
/**
|
|
@@ -3704,6 +3753,8 @@ export interface PipelineAttempt {
|
|
|
3704
3753
|
gpu_seconds: number | null;
|
|
3705
3754
|
/** As the versions are priced once it has ended; as recorded so far while it runs. */
|
|
3706
3755
|
cost_cents: number;
|
|
3756
|
+
/** See {@link PipelineVersion.cost_up_to}. */
|
|
3757
|
+
cost_up_to: boolean;
|
|
3707
3758
|
verdict: TrainingVerdict | null;
|
|
3708
3759
|
win_rate: number | null;
|
|
3709
3760
|
evaluation_id: string | null;
|
|
@@ -3742,8 +3793,6 @@ export interface PipelineTrainWhen {
|
|
|
3742
3793
|
}
|
|
3743
3794
|
/** What is held back beyond max(50, 5% of the data): a larger percent, or a count. */
|
|
3744
3795
|
export interface PipelineHoldout {
|
|
3745
|
-
percent: number;
|
|
3746
|
-
count: number | null;
|
|
3747
3796
|
min_rows: number;
|
|
3748
3797
|
}
|
|
3749
3798
|
/**
|
|
@@ -3780,10 +3829,6 @@ export interface Pipeline {
|
|
|
3780
3829
|
* commit the server pinned.
|
|
3781
3830
|
*/
|
|
3782
3831
|
challenger_model: PipelineModelRef | null;
|
|
3783
|
-
/** `sft` today; `kto`, `dpo` and `grpo` cannot be chosen yet. */
|
|
3784
|
-
training_type: string;
|
|
3785
|
-
/** How each version starts: `fresh`, a new adapter over the pinned base trained on all accepted data. */
|
|
3786
|
-
version_base: string;
|
|
3787
3832
|
train_when: PipelineTrainWhen;
|
|
3788
3833
|
promotion: PipelinePromotion;
|
|
3789
3834
|
holdout: PipelineHoldout;
|
|
@@ -3793,7 +3838,6 @@ export interface Pipeline {
|
|
|
3793
3838
|
/** The pinned base every version is a fresh adapter over, which is what makes their scores comparable. */
|
|
3794
3839
|
base_model_id: string;
|
|
3795
3840
|
base_model_revision: string;
|
|
3796
|
-
train_type: string;
|
|
3797
3841
|
/** The version serving now; null when what serves is not a version of this pipeline. */
|
|
3798
3842
|
live_version: number | null;
|
|
3799
3843
|
/** The same number under its older name. */
|
|
@@ -3808,25 +3852,31 @@ export interface Pipeline {
|
|
|
3808
3852
|
serving_taken_over_by: string | null;
|
|
3809
3853
|
benchmark_id: string | null;
|
|
3810
3854
|
benchmark_name: string | null;
|
|
3811
|
-
|
|
3855
|
+
/** The least a benchmark's score has to rise by to count as an improvement under a policy that reads benchmarks. */
|
|
3812
3856
|
benchmark_min_delta: number;
|
|
3813
3857
|
status: PipelineStatus;
|
|
3814
3858
|
/**
|
|
3815
3859
|
* The rule's own sentence about what it is waiting for or why it stopped,
|
|
3816
3860
|
* verbatim. Only as fresh as the platform's last visit: when `status` is
|
|
3817
|
-
* `paused`, `paused_reason`
|
|
3861
|
+
* `paused`, `paused_reason` is what says why.
|
|
3818
3862
|
*/
|
|
3819
3863
|
next_reason: string | null;
|
|
3820
3864
|
last_checked_at: string | null;
|
|
3821
3865
|
next_due_at: string | null;
|
|
3822
3866
|
auto_promote: boolean;
|
|
3823
|
-
/**
|
|
3867
|
+
/**
|
|
3868
|
+
* Why the platform stopped firing this rule, or null. `consent_invalid` is
|
|
3869
|
+
* a pipeline whose settings changed since it was last saved: saving it
|
|
3870
|
+
* again starts it.
|
|
3871
|
+
*/
|
|
3824
3872
|
paused_reason: TrainingRulePausedReason | null;
|
|
3825
3873
|
/**
|
|
3826
|
-
*
|
|
3827
|
-
*
|
|
3874
|
+
* When this workspace's scoring key comes off its monthly cap (the start of
|
|
3875
|
+
* the next UTC month), or null when it is not capped. Until then Train now
|
|
3876
|
+
* answers 409 `SCORING_CAPPED` with this time as `resumes_at`, and nothing
|
|
3877
|
+
* trains on its own: an attempt would train and could not be compared.
|
|
3828
3878
|
*/
|
|
3829
|
-
|
|
3879
|
+
scoring_capped_until: string | null;
|
|
3830
3880
|
/** The rule's monthly limit. Null means it has none. */
|
|
3831
3881
|
monthly_ceiling_cents: number | null;
|
|
3832
3882
|
/**
|
|
@@ -3834,10 +3884,11 @@ export interface Pipeline {
|
|
|
3834
3884
|
* this UTC calendar month -- the figure the limit is enforced against. An
|
|
3835
3885
|
* upper bound, not an exact spend. A run counts toward the month it was
|
|
3836
3886
|
* created in, and a run still in progress also counts toward the current
|
|
3837
|
-
* month, at
|
|
3838
|
-
*
|
|
3839
|
-
*
|
|
3840
|
-
*
|
|
3887
|
+
* month, at the most it may cost (as `run_max_cents` counts it, a base
|
|
3888
|
+
* model's machine included where the attempt has one of its own) or what it
|
|
3889
|
+
* has been billed when that is more. A finished run counts what it was billed, and its comparison
|
|
3890
|
+
* machines are billed at the most they could have cost (each machine's
|
|
3891
|
+
* hourly cap for the time it was up, never more than its own amount). A run created in
|
|
3841
3892
|
* an earlier month that finishes in this one counts here only while it is
|
|
3842
3893
|
* still running.
|
|
3843
3894
|
*/
|