runbios-sdk 0.2.19-dev.274 → 0.2.19-dev.278

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/dist/index.d.ts CHANGED
@@ -36,7 +36,7 @@ import { Loop } from './resources/loop.js';
36
36
  * SDK version. Sent as part of the User-Agent header.
37
37
  * Must match package.json "version" -- enforced by a contract test.
38
38
  */
39
- export declare const VERSION = "0.2.19-dev.274";
39
+ export declare const VERSION = "0.2.19-dev.278";
40
40
  export declare class RunBiOS {
41
41
  /** Search models, fetch configs, check adapter compatibility. */
42
42
  readonly models: Models;
@@ -75,7 +75,7 @@ export { Training } from './resources/training.js';
75
75
  export { Wallet } from './resources/wallet.js';
76
76
  export { GPU, type GPURecommendation } from './resources/gpu.js';
77
77
  export { Inference, validateChatRequest, parseSSE, buildInferenceRequest, contextSizingBasis, type ContextSizingBasis, CONTEXT_DEFAULT_CEILING, CONTEXT_EDITABLE_FLOOR, type ChatMessage, type FunctionTool, type ChatCompletionParams, type ChatCompletionResponse, type ChatCompletionChunk, type InferenceVerbParams, type CompletionParams, type EmbeddingParams, type RerankParams, type AnthropicMessageParams, type ServerlessUsageWindow, type ServerlessTimeseriesMetric, type ServerlessWorkspaceLimits, type ServerlessUsageEnvelope, type ServerlessUsageOverviewResponse, type ServerlessUsageRowsResponse, type ServerlessUsageTimeseriesResponse, type ServerlessUsageDailyResponse, type ServerlessSavingsTotals, type ServerlessSavingsPeriod, type ServerlessUsageSavingsResponse, } from './resources/inference.js';
78
- export type { BiOSConfig, PaginatedResponse, ApiErrorBody, AvailableGpuAlternative, CapacityMinimumRequirement, InferenceBookingAccepted, Model, ModelSearchParams, ModelSearchResponse, ModelDetailResponse, InferenceModel, InferenceModelListResponse, ModelConfig, Dataset, DatasetListParams, DatasetListResponse, DatasetUploadParams, DatasetPreview, DatasetPreviewParams, DatasetImportHFParams, DatasetRegisterHFParams, DatasetHubSearchParams, DatasetHubPreviewParams, DatasetValidation, DatasetFormatVariant, DatasetFormatSpec, DatasetFormatSpecs, DatasetStorageUsage, TrainingMethod, RLHFAlgorithm, AdapterType, TrainingJobStatus, TrainingStatusFilter, TrainingCreateParams, TrainingListParams, TrainingListResponse, TrainingJob, TrainingMetrics, TrainingResourceMetrics, TrainingDeviceMetrics, MetricPoint, MetricGraphConfig, TrainingCheckpoint, TrainingLogs, TrainingLogEntry, TrainingStopResponse, TrainingResumeResponse, CanonicalTrainingRequest, TrainingPreflightDataset, TrainingPreflightWarning, TrainingPreflightResponse, TrainingCapabilityChoice, TrainingConfigFieldCapability, TrainingCapabilities, GPUChoice, WalletBalance, Transaction, TransactionListResponse, TransactionListParams, GPUInfo, GPUPricingResponse, GPUOptionsParams, TrainingRecommendParams, TrainingAdvisorVerdict, TrainingAdvisorRecommendation, TrainingAdvisorUserShape, TrainingAdvisorJustification, TrainingAdvisorBasis, GPUOption, GPUOptionSuggestion, GPUOptionsResponse, InferenceStatus, InferenceCreateParams, InferenceUpdateParams, InferenceDeployment, InferenceDeploymentSummary, InferenceCreateResponse, InferenceListResponse, InferenceUpdateResponse, InferenceLifecycleResponse, InferenceDeleteResponse, InferenceGPUOptionMarket, InferenceGPUOption, InferenceGPUAlternative, InferenceGPUOptionsParams, InferenceGPUOptionsResponse, InferenceCanonicalRequest, InferencePreflightWarning, InferencePreflightResponse, AdapterCompatibility, AdapterCompatibilityResponse, AdapterCompatibilityParams, SupportedArchitecture, ArchitectureScope, SupportedArchitecturesResponse, ApiKey, ApiKeyScope, ApiKeyIntrospection, Organization, OrgMember, OrgInvite, Workspace, Integration, IntegrationCreateParams, StorageUsage, StorageObject, PromptTokensDetails, ChatCompletionUsage, ServingKind, TrainingRulePausedReason, TrainingTrigger, TrainingRecipe, TrainingRunState, TrainingVerdict, TrainingDecision, NotifyState, EvaluationStatus, EvaluationItemStatus, EvaluationWinner, EvaluationItemWinnerFilter, EvaluationWarningCode, TrainingToolCall, TrainingMessage, TrainingRuleServing, TrainingRunPromoteRequest, TrainingRunRejectRequest, TrainingRunRollbackRequest, TrainingRunCancelRequest, TrainingRunListParams, TrainingRun, TrainingRunSummary, TrainingRunEvent, TrainingRunLinks, TrainingRunActions, EvaluationRef, EvaluationDecoding, EvaluationJudgeDimension, EvaluationDimensionScore, EvaluationWarning, EvaluationMargin, JudgeAgreementDimension, JudgeAgreement, Evaluation, EvaluationItemSide, EvaluationItem, EvaluationItemListParams, BenchmarkStatus, BenchmarkSourceKind, BenchmarkRunStatus, BenchmarkScoring, BenchmarkDimensionScore, Benchmark, BenchmarkItem, BenchmarkRun, BenchmarkHistoryPoint, BenchmarkSource, BenchmarkCreateParams, BenchmarkListParams, BenchmarkItemListParams, BenchmarkHistoryParams, BenchmarkRetireParams, InferenceAlias, InferenceAliasRequest, TrainingRuleDeleteResponse, TrainingRunListResponse, TrainingRunResponse, TrainingRunActionResponse, EvaluationItemsResponse, InferenceAliasListResponse, InferenceAliasResponse, InferenceAliasDeleteResponse, BenchmarkListResponse, BenchmarkItemsResponse, BenchmarkHistoryResponse, TrainingRuleRunRefusalCode, TrainingRuleSaveRefusalCode, TrainingRuleMonthlyLimitRefusal, PipelineStatus, PipelineActiveRun, PipelineVersion, PipelineAttempt, PipelineAttemptTrigger, PipelineAttemptOutcome, Pipeline, PipelineListParams, PipelineListResponse, PipelineResponse, PipelineTrainWhen, PipelineSchedule, PipelineWriteRequest, PipelineMutationResponse, PipelineImportParams, PipelineFloors, LoopModel, LoopModelsResponse, LoopTracePipeline, LoopTracePipelinesResponse, LoopMetricsHistoryParams, LoopMetricPoint, LoopSignalDay, LoopMetricsHistory, PromotionPolicyState, RuleBenchmark, PipelineBenchmarksResponse, PipelineBenchmarkAttachParams, PipelineBenchmarkMoveParams, CompareBenchmark, VersionBenchmarkScore, PromotionMeasurement, PromotionOutcome, VersionScores, VersionScoreDifference, VersionCompareResponse, VersionCompareParams, } from './types.js';
78
+ export type { BiOSConfig, PaginatedResponse, ApiErrorBody, AvailableGpuAlternative, CapacityMinimumRequirement, InferenceBookingAccepted, Model, ModelSearchParams, ModelSearchResponse, ModelDetailResponse, InferenceModel, InferenceModelListResponse, ModelConfig, Dataset, DatasetListParams, DatasetListResponse, DatasetUploadParams, DatasetPreview, DatasetPreviewParams, DatasetImportHFParams, DatasetRegisterHFParams, DatasetHubSearchParams, DatasetHubPreviewParams, DatasetValidation, DatasetFormatVariant, DatasetFormatSpec, DatasetFormatSpecs, DatasetStorageUsage, TrainingMethod, RLHFAlgorithm, AdapterType, TrainingJobStatus, TrainingStatusFilter, TrainingCreateParams, TrainingListParams, TrainingListResponse, TrainingJob, TrainingMetrics, TrainingResourceMetrics, TrainingDeviceMetrics, MetricPoint, MetricGraphConfig, TrainingCheckpoint, TrainingLogs, TrainingLogEntry, TrainingStopResponse, TrainingResumeResponse, CanonicalTrainingRequest, TrainingPreflightDataset, TrainingPreflightWarning, TrainingPreflightResponse, TrainingCapabilityChoice, TrainingConfigFieldCapability, TrainingCapabilities, GPUChoice, WalletBalance, Transaction, TransactionListResponse, TransactionListParams, GPUInfo, GPUPricingResponse, GPUOptionsParams, TrainingRecommendParams, TrainingAdvisorVerdict, TrainingAdvisorRecommendation, TrainingAdvisorUserShape, TrainingAdvisorJustification, TrainingAdvisorBasis, GPUOption, GPUOptionSuggestion, GPUOptionsResponse, InferenceStatus, InferenceCreateParams, InferenceUpdateParams, InferenceDeployment, InferenceDeploymentSummary, InferenceCreateResponse, InferenceListResponse, InferenceUpdateResponse, InferenceLifecycleResponse, InferenceDeleteResponse, InferenceGPUOptionMarket, InferenceGPUOption, InferenceGPUAlternative, InferenceGPUOptionsParams, InferenceGPUOptionsResponse, InferenceCanonicalRequest, InferencePreflightWarning, InferencePreflightResponse, AdapterCompatibility, AdapterCompatibilityResponse, AdapterCompatibilityParams, SupportedArchitecture, ArchitectureScope, SupportedArchitecturesResponse, ApiKey, ApiKeyScope, ApiKeyIntrospection, Organization, OrgMember, OrgInvite, Workspace, Integration, IntegrationCreateParams, StorageUsage, StorageObject, PromptTokensDetails, ChatCompletionUsage, ServingKind, TrainingRulePausedReason, TrainingTrigger, TrainingRecipe, TrainingRunState, TrainingVerdict, TrainingDecision, EvaluationStatus, EvaluationItemStatus, EvaluationWinner, EvaluationItemWinnerFilter, EvaluationWarningCode, TrainingToolCall, TrainingMessage, TrainingRuleServing, TrainingRunPromoteRequest, TrainingRunRejectRequest, TrainingRunRollbackRequest, TrainingRunCancelRequest, TrainingRunListParams, TrainingRun, TrainingRunSummary, TrainingRunEvent, TrainingRunLinks, TrainingRunActions, EvaluationRef, EvaluationDecoding, EvaluationJudgeDimension, EvaluationDimensionScore, EvaluationWarning, EvaluationMargin, Evaluation, EvaluationItemSide, EvaluationItem, EvaluationItemListParams, BenchmarkStatus, BenchmarkSourceKind, BenchmarkRunStatus, BenchmarkScoring, BenchmarkDimensionScore, Benchmark, BenchmarkItem, BenchmarkRun, BenchmarkHistoryPoint, BenchmarkSource, BenchmarkCreateParams, BenchmarkListParams, BenchmarkItemListParams, BenchmarkHistoryParams, BenchmarkRetireParams, InferenceAlias, InferenceAliasRequest, TrainingRuleDeleteResponse, TrainingRunListResponse, TrainingRunResponse, TrainingRunActionResponse, EvaluationItemsResponse, InferenceAliasListResponse, InferenceAliasResponse, InferenceAliasDeleteResponse, BenchmarkListResponse, BenchmarkItemsResponse, BenchmarkHistoryResponse, TrainingRuleRunRefusalCode, TrainingRuleSaveRefusalCode, TrainingRuleMonthlyLimitRefusal, PipelineStatus, PipelineActiveRun, PipelineVersion, PipelineAttempt, PipelineAttemptTrigger, PipelineAttemptOutcome, Pipeline, PipelineListParams, PipelineListResponse, PipelineResponse, PipelineTrainWhen, PipelineSchedule, PipelineWriteRequest, PipelineMutationResponse, PipelineImportParams, PipelineFloors, LoopModel, LoopModelsResponse, LoopTracePipeline, LoopTracePipelinesResponse, LoopMetricsHistoryParams, LoopMetricPoint, LoopSignalDay, LoopMetricsHistory, PromotionPolicyState, RuleBenchmark, PipelineBenchmarksResponse, PipelineBenchmarkAttachParams, PipelineBenchmarkMoveParams, CompareBenchmark, VersionBenchmarkScore, PromotionMeasurement, PromotionOutcome, VersionScores, VersionScoreDifference, VersionCompareResponse, VersionCompareParams, } from './types.js';
79
79
  /** The closed set of run states a training run never leaves. */
80
80
  export { TERMINAL_RUN_STATES } from './types.js';
81
81
  /** The most versions a pipeline may be set to make. */
package/dist/index.js CHANGED
@@ -36,7 +36,7 @@ import { Loop } from './resources/loop.js';
36
36
  * SDK version. Sent as part of the User-Agent header.
37
37
  * Must match package.json "version" -- enforced by a contract test.
38
38
  */
39
- export const VERSION = '0.2.19-dev.274';
39
+ export const VERSION = '0.2.19-dev.278';
40
40
  export class RunBiOS {
41
41
  /** Search models, fetch configs, check adapter compatibility. */
42
42
  models;
@@ -88,10 +88,10 @@ export declare class Loop {
88
88
  * a corpus from a pile.
89
89
  *
90
90
  * Each row is read for what it actually is. A `prompt`/`chosen`/`rejected`
91
- * triple becomes a preference pair; an answer plus a yes-or-no becomes a
92
- * thumbs verdict; a question and an answer waits for review; a question with
93
- * no answer waits for an answer; a paragraph of prose is refused, because it
94
- * is not a conversation.
91
+ * triple becomes a correction (`chosen` is learned from); an answer plus a
92
+ * yes-or-no becomes a thumbs verdict; a question and an answer waits for
93
+ * review; a question with no answer waits for an answer; a paragraph of prose
94
+ * is refused, because it is not a conversation.
95
95
  *
96
96
  * A verdict that arrives *with* the file is kept -- discarding somebody's
97
97
  * judgement would be worse -- but it is recorded as having come from your
@@ -138,7 +138,14 @@ export declare class Loop {
138
138
  importRows(params: LoopImportParams): Promise<LoopImportResult>;
139
139
  /** List captured conversations, newest first. */
140
140
  listTraces(params?: LoopTraceListParams): Promise<LoopTraceListResponse>;
141
- /** Read one conversation, with every verdict recorded on it. */
141
+ /**
142
+ * Read one conversation, with every verdict and tag recorded on it.
143
+ *
144
+ * A conversation kept in your own storage carries `payload_stored_externally`.
145
+ * When its text could not be read just now it also carries
146
+ * `payload_unavailable`, with `messages` null and `completion` empty: it is
147
+ * not an empty conversation. Read it again later rather than reviewing it.
148
+ */
142
149
  getTrace(id: string): Promise<LoopTrace>;
143
150
  /**
144
151
  * Delete a conversation.
@@ -158,11 +165,11 @@ export declare class Loop {
158
165
  * Append-only: posting a second verdict does not replace the first. Two
159
166
  * reviewers disagreeing about an answer is information worth keeping.
160
167
  *
161
- * `source` says who judged: `human`, `verifier` (a deterministic check --
162
- * tests passed, schema valid), `judge` (a model grading a model), or
163
- * `behavioural` (what the user did next). When several disagree, a human
164
- * outranks a verifier, a verifier outranks a judge, and a judge outranks a
165
- * behavioural hint.
168
+ * What a pipeline trains on (every pipeline fine-tunes on examples): an
169
+ * `accepted` answer is an example as it stands; the `correction` of an
170
+ * `edited` or `gold` verdict is the example, in place of the model's answer;
171
+ * a `rejected` answer is left out. Nothing is learned from an answer that was
172
+ * rejected or replaced. `source` defaults to `human`.
166
173
  */
167
174
  signal(traceId: string, params: LoopSignalParams): Promise<LoopSignal>;
168
175
  /**
@@ -182,23 +189,18 @@ export declare class Loop {
182
189
  /**
183
190
  * The model was wrong; here is what it should have said.
184
191
  *
185
- * The most valuable feedback there is, because it produces BOTH halves of a
186
- * preference pair from one action: the model learns your answer and learns to
187
- * avoid its own.
192
+ * The most valuable feedback there is: your answer is trained on in place of
193
+ * the model's. Nothing is learned from the answer it replaced.
188
194
  */
189
195
  correct(traceId: string, answer: string, reason?: string): Promise<LoopSignal>;
190
196
  /**
191
197
  * Record the GOLD answer for this question — the reference, regardless of
192
198
  * what the model happened to say.
193
199
  *
194
- * Different from `correct()` in a way that matters: a correction asserts the
195
- * model was wrong, so the original becomes the rejected half of a pair. A gold
196
- * answer asserts nothing about the model, so if it MATCHES what was said, no
197
- * preference pair is invented — but SFT still learns it.
198
- *
199
- * It is the most reusable thing you can record: it trains SFT, forms a DPO
200
- * pair when it differs from the answer, and stands in as the GRPO reference
201
- * when you have not supplied a separate ground truth.
200
+ * A pipeline trains on it exactly as it does on a correction: the gold
201
+ * answer is the example, in place of the model's. The difference is only
202
+ * what it says: a correction says the model was wrong, a gold answer says
203
+ * nothing about the model.
202
204
  */
203
205
  gold(traceId: string, answer: string, reason?: string): Promise<LoopSignal>;
204
206
  /** Every verdict on one conversation, oldest first. */
@@ -356,8 +358,8 @@ export declare class Loop {
356
358
  *
357
359
  * `status` says whether it is the platform working (`running`), the member
358
360
  * who has to act (`needs_review`, `needs_funds`), or nothing at all until
359
- * somebody does (`paused`, with `paused_reason` or `needs_consent` saying
360
- * why; `complete` at `max_versions`). `month_spent_cents` is the figure the
361
+ * somebody does (`paused`, with `paused_reason` saying why; `complete` at
362
+ * `max_versions`). `month_spent_cents` is the figure the
361
363
  * monthly limit is enforced against: what runs started this month have cost
362
364
  * at most, not an exact spend.
363
365
  */
@@ -91,10 +91,10 @@ export class Loop {
91
91
  * a corpus from a pile.
92
92
  *
93
93
  * Each row is read for what it actually is. A `prompt`/`chosen`/`rejected`
94
- * triple becomes a preference pair; an answer plus a yes-or-no becomes a
95
- * thumbs verdict; a question and an answer waits for review; a question with
96
- * no answer waits for an answer; a paragraph of prose is refused, because it
97
- * is not a conversation.
94
+ * triple becomes a correction (`chosen` is learned from); an answer plus a
95
+ * yes-or-no becomes a thumbs verdict; a question and an answer waits for
96
+ * review; a question with no answer waits for an answer; a paragraph of prose
97
+ * is refused, because it is not a conversation.
98
98
  *
99
99
  * A verdict that arrives *with* the file is kept -- discarding somebody's
100
100
  * judgement would be worse -- but it is recorded as having come from your
@@ -182,7 +182,14 @@ export class Loop {
182
182
  const qs = q.toString();
183
183
  return this._http.fetchGet(`/api/loop/traces${qs ? `?${qs}` : ''}`);
184
184
  }
185
- /** Read one conversation, with every verdict recorded on it. */
185
+ /**
186
+ * Read one conversation, with every verdict and tag recorded on it.
187
+ *
188
+ * A conversation kept in your own storage carries `payload_stored_externally`.
189
+ * When its text could not be read just now it also carries
190
+ * `payload_unavailable`, with `messages` null and `completion` empty: it is
191
+ * not an empty conversation. Read it again later rather than reviewing it.
192
+ */
186
193
  async getTrace(id) {
187
194
  const res = await this._http.fetchGet(`/api/loop/traces/${encodeURIComponent(id)}`);
188
195
  return res.trace;
@@ -205,11 +212,11 @@ export class Loop {
205
212
  * Append-only: posting a second verdict does not replace the first. Two
206
213
  * reviewers disagreeing about an answer is information worth keeping.
207
214
  *
208
- * `source` says who judged: `human`, `verifier` (a deterministic check --
209
- * tests passed, schema valid), `judge` (a model grading a model), or
210
- * `behavioural` (what the user did next). When several disagree, a human
211
- * outranks a verifier, a verifier outranks a judge, and a judge outranks a
212
- * behavioural hint.
215
+ * What a pipeline trains on (every pipeline fine-tunes on examples): an
216
+ * `accepted` answer is an example as it stands; the `correction` of an
217
+ * `edited` or `gold` verdict is the example, in place of the model's answer;
218
+ * a `rejected` answer is left out. Nothing is learned from an answer that was
219
+ * rejected or replaced. `source` defaults to `human`.
213
220
  */
214
221
  async signal(traceId, params) {
215
222
  const res = await this._http.fetchPost(`/api/loop/traces/${encodeURIComponent(traceId)}/signals`, params);
@@ -236,9 +243,8 @@ export class Loop {
236
243
  /**
237
244
  * The model was wrong; here is what it should have said.
238
245
  *
239
- * The most valuable feedback there is, because it produces BOTH halves of a
240
- * preference pair from one action: the model learns your answer and learns to
241
- * avoid its own.
246
+ * The most valuable feedback there is: your answer is trained on in place of
247
+ * the model's. Nothing is learned from the answer it replaced.
242
248
  */
243
249
  async correct(traceId, answer, reason) {
244
250
  return this.signal(traceId, { verdict: 'edited', correction: answer, reason });
@@ -247,14 +253,10 @@ export class Loop {
247
253
  * Record the GOLD answer for this question — the reference, regardless of
248
254
  * what the model happened to say.
249
255
  *
250
- * Different from `correct()` in a way that matters: a correction asserts the
251
- * model was wrong, so the original becomes the rejected half of a pair. A gold
252
- * answer asserts nothing about the model, so if it MATCHES what was said, no
253
- * preference pair is invented — but SFT still learns it.
254
- *
255
- * It is the most reusable thing you can record: it trains SFT, forms a DPO
256
- * pair when it differs from the answer, and stands in as the GRPO reference
257
- * when you have not supplied a separate ground truth.
256
+ * A pipeline trains on it exactly as it does on a correction: the gold
257
+ * answer is the example, in place of the model's. The difference is only
258
+ * what it says: a correction says the model was wrong, a gold answer says
259
+ * nothing about the model.
258
260
  */
259
261
  async gold(traceId, answer, reason) {
260
262
  return this.signal(traceId, { verdict: 'gold', correction: answer, reason });
@@ -491,8 +493,8 @@ export class Loop {
491
493
  *
492
494
  * `status` says whether it is the platform working (`running`), the member
493
495
  * who has to act (`needs_review`, `needs_funds`), or nothing at all until
494
- * somebody does (`paused`, with `paused_reason` or `needs_consent` saying
495
- * why; `complete` at `max_versions`). `month_spent_cents` is the figure the
496
+ * somebody does (`paused`, with `paused_reason` saying why; `complete` at
497
+ * `max_versions`). `month_spent_cents` is the figure the
496
498
  * monthly limit is enforced against: what runs started this month have cost
497
499
  * at most, not an exact spend.
498
500
  */
package/dist/types.d.ts CHANGED
@@ -2297,18 +2297,17 @@ export interface ChatCompletionUsage {
2297
2297
  total_tokens: number;
2298
2298
  prompt_tokens_details?: PromptTokensDetails;
2299
2299
  }
2300
- /** What a training set is shaped for. The three need different things. */
2301
- export type LoopMethod = 'sft' | 'dpo' | 'grpo' | 'kto';
2302
- /** Who judged an answer. */
2300
+ /** Who gave a verdict. Feedback you send is `human` unless you say otherwise. */
2303
2301
  export type LoopSource = 'human' | 'verifier' | 'judge' | 'behavioural';
2304
- /** The judgement itself. */
2305
2302
  /**
2306
2303
  * The judgement recorded on an answer.
2307
2304
  *
2308
- * `accepted` / `rejected` are the thumbs (see `upvote` / `downvote`).
2309
- * `edited` says the model was WRONG and carries the better answer.
2305
+ * `accepted` / `rejected` are the thumbs (see `upvote` / `downvote`): an
2306
+ * accepted answer is trained on, a rejected one is left out.
2307
+ * `edited` says the model was WRONG and carries the better answer, which is
2308
+ * trained on in place of the model's.
2310
2309
  * `gold` records the reference answer for the question, making no claim about
2311
- * whether the model was right — which is why it is separate from `edited`.
2310
+ * whether the model was right; it is trained on the same way.
2312
2311
  */
2313
2312
  export type LoopVerdict = 'accepted' | 'rejected' | 'edited' | 'scored' | 'gold';
2314
2313
  /** A function an assistant turn asked to invoke. `arguments` is JSON-encoded. */
@@ -2495,7 +2494,7 @@ export interface LoopImportResult {
2495
2494
  not_saved_rows?: number[];
2496
2495
  /**
2497
2496
  * Rows whose CONVERSATION stored and whose VERDICT did not, and which rows
2498
- * those were. A preference pair or a thumbs label is two writes, and the
2497
+ * those were. A chosen/rejected row or a thumbs label is two writes, and the
2499
2498
  * second can fail on its own: the conversation is then in the loop carrying
2500
2499
  * nobody's judgement, counted under `needs_review` rather than `reviewed`,
2501
2500
  * and not trainable. The call rejects with {@link LoopImportIncompleteError},
@@ -2532,6 +2531,8 @@ export interface LoopSignal {
2532
2531
  ground_truth: string | null;
2533
2532
  reason: string | null;
2534
2533
  author: string | null;
2534
+ /** What the caller attached when recording it, or null. */
2535
+ metadata: Record<string, unknown> | null;
2535
2536
  created_at: string;
2536
2537
  }
2537
2538
  export interface LoopSignalParams {
@@ -2539,9 +2540,9 @@ export interface LoopSignalParams {
2539
2540
  source?: LoopSource;
2540
2541
  /** Required when verdict is `edited`. The answer the model should have given. */
2541
2542
  correction?: string;
2542
- /** Required when verdict is `scored`. */
2543
+ /** Required when verdict is `scored`: above zero trains the answer as it stands, otherwise it is left out. */
2543
2544
  score?: number;
2544
- /** A value or fact the answer can be checked against. GRPO needs one. */
2545
+ /** A value or fact the answer can be checked against. Kept with the verdict; a pipeline trains on `correction`, not on this. */
2545
2546
  ground_truth?: string;
2546
2547
  reason?: string;
2547
2548
  author?: string;
@@ -2579,7 +2580,11 @@ export interface LoopTrace {
2579
2580
  deployment_id: string | null;
2580
2581
  model: string;
2581
2582
  model_version: string | null;
2582
- messages: LoopMessage[];
2583
+ /**
2584
+ * The conversation. NULL, with `completion` empty, when its text is kept in
2585
+ * your own storage and could not be read just now: see `payload_unavailable`.
2586
+ */
2587
+ messages: LoopMessage[] | null;
2583
2588
  completion: string;
2584
2589
  tool_calls?: LoopToolCall[];
2585
2590
  tools?: unknown;
@@ -2591,9 +2596,40 @@ export interface LoopTrace {
2591
2596
  redacted_at: string | null;
2592
2597
  /** What the redaction pass removed, counted by rule. */
2593
2598
  redaction_report?: Record<string, number>;
2599
+ /** What the caller attached when recording it, or null. */
2600
+ metadata: Record<string, unknown> | null;
2594
2601
  created_at: string;
2595
2602
  expires_at: string;
2603
+ /** Every verdict recorded on it, oldest first. On the single read only. */
2596
2604
  signals?: LoopSignal[];
2605
+ /** How many verdicts it carries. */
2606
+ signal_count: number;
2607
+ /**
2608
+ * The verdict that decides it -- the one training uses, not whichever
2609
+ * arrived last. Absent when nobody has given feedback.
2610
+ */
2611
+ verdict?: LoopVerdict;
2612
+ /** Who gave that verdict. */
2613
+ verdict_source?: LoopSource;
2614
+ /**
2615
+ * The score a `scored` verdict carries: above zero it trains the answer as
2616
+ * it stands (good feedback), zero or below it trains nothing (bad).
2617
+ */
2618
+ verdict_score?: number;
2619
+ /** Its tags. On the single read only. */
2620
+ labels?: LoopLabel[];
2621
+ /**
2622
+ * True when the conversation's text is kept in your own storage rather than
2623
+ * here. The text is read from there on every request.
2624
+ */
2625
+ payload_stored_externally?: boolean;
2626
+ /**
2627
+ * True when that text could not be read just now (the storage could not be
2628
+ * reached, or the object is gone). `messages` is then null and `completion`
2629
+ * empty: the conversation is NOT empty. Try again later; do not review it or
2630
+ * treat it as a blank answer.
2631
+ */
2632
+ payload_unavailable?: boolean;
2597
2633
  }
2598
2634
  export interface LoopTraceListParams {
2599
2635
  deployment_id?: string;
@@ -2810,11 +2846,13 @@ export type TrainingRulePausedReason = 'consent_invalid' | 'insufficient_funds'
2810
2846
  */
2811
2847
  | 'training_not_enabled';
2812
2848
  /**
2813
- * Why a run fired. `variant` is a recipe variant: the pipeline had nothing new
2814
- * to learn, so it trained the same base model on the same conversations with
2815
- * one training setting changed (see {@link TrainingRecipe}).
2849
+ * Why a run fired, as the run stores it: `rows` (enough new samples),
2850
+ * `cadence` (the schedule), `both`, `manual` (train now), `challenger` (the
2851
+ * pipeline's challenger model trained on the same data as the attempt before
2852
+ * it), or `variant` (a recipe variant, retired: the same base model on the
2853
+ * same conversations with one training setting changed; old runs only).
2816
2854
  */
2817
- export type TrainingTrigger = 'rows' | 'cadence' | 'both' | 'manual' | 'variant';
2855
+ export type TrainingTrigger = 'rows' | 'cadence' | 'both' | 'manual' | 'variant' | 'challenger';
2818
2856
  /**
2819
2857
  * The four training settings a recipe variant may change, as the run was
2820
2858
  * actually trained: the training service's own defaults are filled in where
@@ -2833,7 +2871,6 @@ export type TrainingRunState = 'built' | 'registering' | 'dataset_validating' |
2833
2871
  export declare const TERMINAL_RUN_STATES: readonly TrainingRunState[];
2834
2872
  export type TrainingVerdict = 'better' | 'not_better' | 'inconclusive' | 'not_evaluated';
2835
2873
  export type TrainingDecision = 'auto_promoted' | 'promoted' | 'rejected' | 'auto_rejected' | 'rolled_back';
2836
- export type NotifyState = 'none' | 'pending' | 'sending' | 'sent' | 'suppressed' | 'failed';
2837
2874
  export type EvaluationStatus = 'open' | 'done' | 'failed' | 'budget_stopped';
2838
2875
  export type EvaluationItemStatus = 'pending' | 'working' | 'done' | 'failed';
2839
2876
  export type EvaluationWinner = 'candidate' | 'incumbent' | 'tie';
@@ -2891,8 +2928,6 @@ export interface TrainingRun {
2891
2928
  workspace_id: string;
2892
2929
  seq: number;
2893
2930
  rule_revision: number;
2894
- authorized_by: string;
2895
- rule_snapshot: Record<string, unknown>;
2896
2931
  trigger: TrainingTrigger;
2897
2932
  /**
2898
2933
  * For a recipe variant, what it changed, in plain words ("half the learning
@@ -2921,13 +2956,9 @@ export interface TrainingRun {
2921
2956
  * no remedy worth printing.
2922
2957
  */
2923
2958
  error_next_step: string | null;
2924
- training_ceiling_cents: number;
2925
- candidate_ceiling_cents: number;
2926
- eval_ceiling_cents: number;
2927
2959
  billed_training_cents: number;
2928
2960
  billed_candidate_cents: number;
2929
2961
  spent_eval_cents: number;
2930
- loop_dataset_id: string | null;
2931
2962
  train_rows: number | null;
2932
2963
  holdout_rows: number | null;
2933
2964
  train_dataset_id: string | null;
@@ -2967,8 +2998,7 @@ export interface TrainingRun {
2967
2998
  benchmark_run_id: string | null;
2968
2999
  /**
2969
3000
  * What the replay's calls cost. Kept apart from `spent_eval_cents` so a
2970
- * report can say what the comparison cost and what the benchmark cost; both
2971
- * come out of `eval_ceiling_cents` and their sum can never exceed it, so
3001
+ * report can say what the comparison cost and what the benchmark cost, so
2972
3002
  * anything totalling what a run cost has to add this one too.
2973
3003
  */
2974
3004
  benchmark_spent_cents: number;
@@ -2982,10 +3012,6 @@ export interface TrainingRun {
2982
3012
  decided_at: string | null;
2983
3013
  auto_promote_at: string | null;
2984
3014
  review_deadline_at: string | null;
2985
- notify_state: NotifyState;
2986
- notify_event: string | null;
2987
- notify_error: string | null;
2988
- notify_attempts: number;
2989
3015
  created_at: string;
2990
3016
  updated_at: string;
2991
3017
  finished_at: string | null;
@@ -3031,7 +3057,6 @@ export interface TrainingRunSummary {
3031
3057
  evaluation_id: string | null;
3032
3058
  review_deadline_at: string | null;
3033
3059
  auto_promote_at: string | null;
3034
- notify_state: NotifyState;
3035
3060
  created_at: string;
3036
3061
  finished_at: string | null;
3037
3062
  }
@@ -3049,7 +3074,6 @@ export interface TrainingRunEvent {
3049
3074
  export interface TrainingRunLinks {
3050
3075
  training_job_url: string | null;
3051
3076
  candidate_url: string | null;
3052
- dataset_url: string | null;
3053
3077
  evaluation_url: string | null;
3054
3078
  /** The TREND the benchmark number belongs to, not the one replay. */
3055
3079
  benchmark_url: string | null;
@@ -3091,41 +3115,13 @@ export interface EvaluationWarning {
3091
3115
  export interface EvaluationMargin {
3092
3116
  promote_margin: number;
3093
3117
  promote_min_win_rate: number;
3094
- min_judge_agreement: number;
3095
3118
  min_holdout_rows: number;
3096
3119
  }
3097
- /**
3098
- * One rubric dimension of a judge-versus-human agreement.
3099
- *
3100
- * Deliberately not EvaluationDimensionScore: that type carries `incumbent` and
3101
- * `candidate`, which are the two models being compared, and an agreement has
3102
- * neither. `pairs` is per dimension because a reviewer who rated one dimension
3103
- * and skipped another leaves a different denominator behind each number.
3104
- */
3105
- export interface JudgeAgreementDimension {
3106
- dimension: string;
3107
- pairs: number;
3108
- agreement: number | null;
3109
- mean_abs_error: number | null;
3110
- }
3111
- /** How often this judge agreed with the workspace's own reviewers. */
3112
- export interface JudgeAgreement {
3113
- judge_id: string;
3114
- pairs: number;
3115
- agreement: number | null;
3116
- mean_abs_error: number | null;
3117
- per_dimension: JudgeAgreementDimension[];
3118
- window_days: number;
3119
- computed_at: string;
3120
- /** "Not enough reviewer overlap yet" is an answer; 100% of two is not. */
3121
- enough_pairs: boolean;
3122
- }
3123
3120
  /** The comparison report: the candidate against what serves today, same rows. */
3124
3121
  export interface Evaluation {
3125
3122
  id: string;
3126
3123
  run_id: string;
3127
3124
  workspace_id: string;
3128
- loop_dataset_id: string | null;
3129
3125
  split: string;
3130
3126
  incumbent_ref: EvaluationRef;
3131
3127
  candidate_ref: EvaluationRef;
@@ -3134,7 +3130,6 @@ export interface Evaluation {
3134
3130
  judge_model: string | null;
3135
3131
  judge_instructions: string | null;
3136
3132
  judge_dimensions: EvaluationJudgeDimension[];
3137
- judge_system_prompt: string | null;
3138
3133
  decoding: EvaluationDecoding;
3139
3134
  rows_selected: number;
3140
3135
  rows_scored: number;
@@ -3150,7 +3145,6 @@ export interface Evaluation {
3150
3145
  judge_incumbent_mean: number | null;
3151
3146
  judge_candidate_mean: number | null;
3152
3147
  per_dimension: EvaluationDimensionScore[];
3153
- judge_agreement: JudgeAgreement | null;
3154
3148
  /** Informational: there is no incumbent counterpart to compare it against. */
3155
3149
  trainer_eval_loss: number | null;
3156
3150
  warnings: EvaluationWarning[];
@@ -3283,14 +3277,11 @@ export interface Benchmark {
3283
3277
  judge_model: string;
3284
3278
  judge_instructions: string;
3285
3279
  judge_dimensions: EvaluationJudgeDimension[];
3286
- judge_system_prompt: string | null;
3287
3280
  decoding: EvaluationDecoding;
3288
3281
  scoring: BenchmarkScoring;
3289
3282
  item_count: number;
3290
3283
  /** Recomputed from the rows and compared at the start of every replay. */
3291
3284
  items_digest: string;
3292
- /** A cap INSIDE the run's own amount for judge calls, never an addition to it. */
3293
- per_run_ceiling_cents: number;
3294
3285
  created_by: string;
3295
3286
  created_at: string;
3296
3287
  retired_at: string | null;
@@ -3342,7 +3333,6 @@ export interface BenchmarkRun {
3342
3333
  prompt_tokens: number;
3343
3334
  completion_tokens: number;
3344
3335
  spent_cents: number;
3345
- ceiling_cents: number;
3346
3336
  status: BenchmarkRunStatus;
3347
3337
  /** The whole explanation when there is no score, so never empty on one. */
3348
3338
  status_reason: string | null;
@@ -3448,6 +3438,10 @@ export interface BenchmarkRetireParams {
3448
3438
  * monthly cap, so what this attempt trained could not be scored. The body
3449
3439
  * carries `resumes_at`, when the cap lifts (the next UTC month, or sooner
3450
3440
  * if the key's cap is raised).
3441
+ * - `LIVE_VERSION_STOPPED` (409): the deployment the pipeline's live version
3442
+ * answers from is stopped, paused until funds are added, deleted or failed,
3443
+ * so an attempt could not be compared against it. Resume it in Deployments,
3444
+ * or roll the version back, then train again.
3451
3445
  * - `AGENT_COULD_NOT_START` (503): the key every comparison is scored with is
3452
3446
  * missing or was refused, and train now could not renew it just then.
3453
3447
  * Nothing was started. Press train now again in a minute.
@@ -3460,7 +3454,7 @@ export interface BenchmarkRetireParams {
3460
3454
  * `MONTHLY_LIMIT_REACHED` and `SCORING_CAPPED` once their date has passed;
3461
3455
  * retrying any other one unchanged gets the same answer.
3462
3456
  */
3463
- export type TrainingRuleRunRefusalCode = 'NOT_ENOUGH_SAMPLES' | 'PIPELINE_NEEDS_SAVE' | 'RULE_PAUSED' | 'RUN_ACTIVE' | 'VERSION_LIMIT_REACHED' | 'MONTHLY_LIMIT_REACHED' | 'SCORING_CAPPED' | 'AGENT_COULD_NOT_START' | 'AGENT_OFF';
3457
+ export type TrainingRuleRunRefusalCode = 'NOT_ENOUGH_SAMPLES' | 'PIPELINE_NEEDS_SAVE' | 'RULE_PAUSED' | 'RUN_ACTIVE' | 'VERSION_LIMIT_REACHED' | 'MONTHLY_LIMIT_REACHED' | 'SCORING_CAPPED' | 'LIVE_VERSION_STOPPED' | 'AGENT_COULD_NOT_START' | 'AGENT_OFF';
3464
3458
  /**
3465
3459
  * The machine codes saving a pipeline (`createPipeline` or `updatePipeline`)
3466
3460
  * can be refused with: every one the two routes answer beyond a malformed
@@ -3520,8 +3514,8 @@ export type TrainingRuleSaveRefusalCode = 'PIPELINE_NAME_TAKEN' | 'OWN_TAG_TAKEN
3520
3514
  *
3521
3515
  * The test is the worst case, not the average: a run may start only if the
3522
3516
  * most the rule's runs this UTC calendar month can have cost
3523
- * (`month_spent_cents`), plus the most the run about to start may spend (its
3524
- * three ceilings added up), fits under the limit.
3517
+ * (`month_spent_cents`), plus the most the run about to start may spend
3518
+ * (`run_max_cents`), fits under the limit.
3525
3519
  */
3526
3520
  export interface TrainingRuleMonthlyLimitRefusal {
3527
3521
  error: {
@@ -3532,12 +3526,17 @@ export interface TrainingRuleMonthlyLimitRefusal {
3532
3526
  month_spent_cents: number;
3533
3527
  /** The rule's monthly limit. */
3534
3528
  monthly_ceiling_cents: number;
3535
- /** The most one run of this rule may spend: training + candidate + evaluation ceilings. */
3529
+ /**
3530
+ * The most one attempt of this rule may spend: its training, its comparison
3531
+ * machine and its judging, and -- on a pipeline, while no version is live --
3532
+ * the base model's own comparison machine beside the new model's.
3533
+ */
3536
3534
  run_max_cents: number;
3537
3535
  /**
3538
3536
  * The first instant of the next UTC month, when the counted spend starts
3539
3537
  * again from nothing. When `run_max_cents` alone exceeds the limit, no month
3540
- * will ever fit and only raising the limit helps; `message` says which.
3538
+ * will ever fit: saving the pipeline sets the limit again from its model's
3539
+ * prices, and nothing else changes it; `message` says which.
3541
3540
  */
3542
3541
  resumes_at: string;
3543
3542
  }
@@ -3549,8 +3548,9 @@ export declare const PIPELINE_MAX_VERSIONS_CEILING = 100;
3549
3548
  * platform working on it.
3550
3549
  *
3551
3550
  * `paused` is a rule that is switched off, AND an enabled rule the platform
3552
- * has stopped firing (`paused_reason`) or whose terms changed and were never
3553
- * accepted (`needs_consent`). `complete` is a pipeline at its `max_versions`.
3551
+ * has stopped firing (`paused_reason`; `consent_invalid` when its settings
3552
+ * changed since it was last saved). `complete` is a pipeline at its
3553
+ * `max_versions`.
3554
3554
  */
3555
3555
  export type PipelineStatus = 'running' | 'needs_review' | 'needs_funds' | 'waiting' | 'paused' | 'complete';
3556
3556
  /**
@@ -3588,8 +3588,6 @@ export interface PipelineActiveRun {
3588
3588
  * not a version, and becomes one only if a member promotes it.
3589
3589
  */
3590
3590
  competed: boolean;
3591
- /** What recipe it is trying, when it is a recipe variant. See TrainingRun.recipe_note. */
3592
- recipe_note: string | null;
3593
3591
  }
3594
3592
  /**
3595
3593
  * One version: an attempt that was put live -- promoted by the pipeline's
@@ -3659,18 +3657,10 @@ export interface PipelineVersion {
3659
3657
  */
3660
3658
  compared_at: string | null;
3661
3659
  /**
3662
- * Why the run that made it fired. `variant` is a version trained on the same
3663
- * conversations as the one before it with one setting changed; every other
3664
- * value is a version that learned from newly reviewed conversations (or a
3665
- * member's run now).
3660
+ * Why the attempt that made it was made, in the words `attempts[]` uses for
3661
+ * the same run; see {@link PipelineAttempt.trigger}.
3666
3662
  */
3667
- trigger: TrainingTrigger;
3668
- /** See TrainingRun.recipe_note. */
3669
- recipe_note: string | null;
3670
- /** See TrainingRun.recipe_of_version. */
3671
- recipe_of_version: number | null;
3672
- /** See TrainingRun.recipe. */
3673
- recipe: TrainingRecipe;
3663
+ trigger: PipelineAttemptTrigger;
3674
3664
  finished_at: string | null;
3675
3665
  }
3676
3666
  /**
@@ -3742,8 +3732,6 @@ export interface PipelineTrainWhen {
3742
3732
  }
3743
3733
  /** What is held back beyond max(50, 5% of the data): a larger percent, or a count. */
3744
3734
  export interface PipelineHoldout {
3745
- percent: number;
3746
- count: number | null;
3747
3735
  min_rows: number;
3748
3736
  }
3749
3737
  /**
@@ -3780,10 +3768,6 @@ export interface Pipeline {
3780
3768
  * commit the server pinned.
3781
3769
  */
3782
3770
  challenger_model: PipelineModelRef | null;
3783
- /** `sft` today; `kto`, `dpo` and `grpo` cannot be chosen yet. */
3784
- training_type: string;
3785
- /** How each version starts: `fresh`, a new adapter over the pinned base trained on all accepted data. */
3786
- version_base: string;
3787
3771
  train_when: PipelineTrainWhen;
3788
3772
  promotion: PipelinePromotion;
3789
3773
  holdout: PipelineHoldout;
@@ -3793,7 +3777,6 @@ export interface Pipeline {
3793
3777
  /** The pinned base every version is a fresh adapter over, which is what makes their scores comparable. */
3794
3778
  base_model_id: string;
3795
3779
  base_model_revision: string;
3796
- train_type: string;
3797
3780
  /** The version serving now; null when what serves is not a version of this pipeline. */
3798
3781
  live_version: number | null;
3799
3782
  /** The same number under its older name. */
@@ -3808,25 +3791,31 @@ export interface Pipeline {
3808
3791
  serving_taken_over_by: string | null;
3809
3792
  benchmark_id: string | null;
3810
3793
  benchmark_name: string | null;
3811
- benchmark_decides: boolean;
3794
+ /** The least a benchmark's score has to rise by to count as an improvement under a policy that reads benchmarks. */
3812
3795
  benchmark_min_delta: number;
3813
3796
  status: PipelineStatus;
3814
3797
  /**
3815
3798
  * The rule's own sentence about what it is waiting for or why it stopped,
3816
3799
  * verbatim. Only as fresh as the platform's last visit: when `status` is
3817
- * `paused`, `paused_reason` and `needs_consent` are what say why.
3800
+ * `paused`, `paused_reason` is what says why.
3818
3801
  */
3819
3802
  next_reason: string | null;
3820
3803
  last_checked_at: string | null;
3821
3804
  next_due_at: string | null;
3822
3805
  auto_promote: boolean;
3823
- /** Why the platform stopped firing this rule, or null. Same codes as the rule's. */
3806
+ /**
3807
+ * Why the platform stopped firing this rule, or null. `consent_invalid` is
3808
+ * a pipeline whose settings changed since it was last saved: saving it
3809
+ * again starts it.
3810
+ */
3824
3811
  paused_reason: TrainingRulePausedReason | null;
3825
3812
  /**
3826
- * The rule's terms changed and nobody has accepted them. It makes nothing
3827
- * until somebody does.
3813
+ * When this workspace's scoring key comes off its monthly cap (the start of
3814
+ * the next UTC month), or null when it is not capped. Until then Train now
3815
+ * answers 409 `SCORING_CAPPED` with this time as `resumes_at`, and nothing
3816
+ * trains on its own: an attempt would train and could not be compared.
3828
3817
  */
3829
- needs_consent: boolean;
3818
+ scoring_capped_until: string | null;
3830
3819
  /** The rule's monthly limit. Null means it has none. */
3831
3820
  monthly_ceiling_cents: number | null;
3832
3821
  /**
@@ -3834,10 +3823,11 @@ export interface Pipeline {
3834
3823
  * this UTC calendar month -- the figure the limit is enforced against. An
3835
3824
  * upper bound, not an exact spend. A run counts toward the month it was
3836
3825
  * created in, and a run still in progress also counts toward the current
3837
- * month, at its three ceilings or what it has been billed when that is
3838
- * more. A finished run counts what it was billed, and its comparison
3839
- * machine is billed at the most it could have cost (the hourly cap for the
3840
- * time it was up, never more than the candidate ceiling). A run created in
3826
+ * month, at the most it may cost (as `run_max_cents` counts it, the base
3827
+ * model's comparison machine included) or what it has been billed when that
3828
+ * is more. A finished run counts what it was billed, and its comparison
3829
+ * machines are billed at the most they could have cost (each machine's
3830
+ * hourly cap for the time it was up, never more than its own amount). A run created in
3841
3831
  * an earlier month that finishes in this one counts here only while it is
3842
3832
  * still running.
3843
3833
  */
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "runbios-sdk",
3
- "version": "0.2.19-dev.274",
3
+ "version": "0.2.19-dev.278",
4
4
  "description": "Official TypeScript SDK for the Run BiOS training and deployment platform API",
5
5
  "type": "module",
6
6
  "main": "./dist/index.js",