runbios-sdk 0.2.19-dev.277 → 0.2.19-dev.281

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/dist/index.d.ts CHANGED
@@ -36,7 +36,7 @@ import { Loop } from './resources/loop.js';
36
36
  * SDK version. Sent as part of the User-Agent header.
37
37
  * Must match package.json "version" -- enforced by a contract test.
38
38
  */
39
- export declare const VERSION = "0.2.19-dev.277";
39
+ export declare const VERSION = "0.2.19-dev.281";
40
40
  export declare class RunBiOS {
41
41
  /** Search models, fetch configs, check adapter compatibility. */
42
42
  readonly models: Models;
@@ -75,7 +75,7 @@ export { Training } from './resources/training.js';
75
75
  export { Wallet } from './resources/wallet.js';
76
76
  export { GPU, type GPURecommendation } from './resources/gpu.js';
77
77
  export { Inference, validateChatRequest, parseSSE, buildInferenceRequest, contextSizingBasis, type ContextSizingBasis, CONTEXT_DEFAULT_CEILING, CONTEXT_EDITABLE_FLOOR, type ChatMessage, type FunctionTool, type ChatCompletionParams, type ChatCompletionResponse, type ChatCompletionChunk, type InferenceVerbParams, type CompletionParams, type EmbeddingParams, type RerankParams, type AnthropicMessageParams, type ServerlessUsageWindow, type ServerlessTimeseriesMetric, type ServerlessWorkspaceLimits, type ServerlessUsageEnvelope, type ServerlessUsageOverviewResponse, type ServerlessUsageRowsResponse, type ServerlessUsageTimeseriesResponse, type ServerlessUsageDailyResponse, type ServerlessSavingsTotals, type ServerlessSavingsPeriod, type ServerlessUsageSavingsResponse, } from './resources/inference.js';
78
- export type { BiOSConfig, PaginatedResponse, ApiErrorBody, AvailableGpuAlternative, CapacityMinimumRequirement, InferenceBookingAccepted, Model, ModelSearchParams, ModelSearchResponse, ModelDetailResponse, InferenceModel, InferenceModelListResponse, ModelConfig, Dataset, DatasetListParams, DatasetListResponse, DatasetUploadParams, DatasetPreview, DatasetPreviewParams, DatasetImportHFParams, DatasetRegisterHFParams, DatasetHubSearchParams, DatasetHubPreviewParams, DatasetValidation, DatasetFormatVariant, DatasetFormatSpec, DatasetFormatSpecs, DatasetStorageUsage, TrainingMethod, RLHFAlgorithm, AdapterType, TrainingJobStatus, TrainingStatusFilter, TrainingCreateParams, TrainingListParams, TrainingListResponse, TrainingJob, TrainingMetrics, TrainingResourceMetrics, TrainingDeviceMetrics, MetricPoint, MetricGraphConfig, TrainingCheckpoint, TrainingLogs, TrainingLogEntry, TrainingStopResponse, TrainingResumeResponse, CanonicalTrainingRequest, TrainingPreflightDataset, TrainingPreflightWarning, TrainingPreflightResponse, TrainingCapabilityChoice, TrainingConfigFieldCapability, TrainingCapabilities, GPUChoice, WalletBalance, Transaction, TransactionListResponse, TransactionListParams, GPUInfo, GPUPricingResponse, GPUOptionsParams, TrainingRecommendParams, TrainingAdvisorVerdict, TrainingAdvisorRecommendation, TrainingAdvisorUserShape, TrainingAdvisorJustification, TrainingAdvisorBasis, GPUOption, GPUOptionSuggestion, GPUOptionsResponse, InferenceStatus, InferenceCreateParams, InferenceUpdateParams, InferenceDeployment, InferenceDeploymentSummary, InferenceCreateResponse, InferenceListResponse, InferenceUpdateResponse, InferenceLifecycleResponse, InferenceDeleteResponse, InferenceGPUOptionMarket, InferenceGPUOption, InferenceGPUAlternative, InferenceGPUOptionsParams, InferenceGPUOptionsResponse, InferenceCanonicalRequest, InferencePreflightWarning, InferencePreflightResponse, AdapterCompatibility, AdapterCompatibilityResponse, AdapterCompatibilityParams, SupportedArchitecture, ArchitectureScope, SupportedArchitecturesResponse, ApiKey, ApiKeyScope, ApiKeyIntrospection, Organization, OrgMember, OrgInvite, Workspace, Integration, IntegrationCreateParams, StorageUsage, StorageObject, PromptTokensDetails, ChatCompletionUsage, ServingKind, TrainingRulePausedReason, TrainingTrigger, TrainingRecipe, TrainingRunState, TrainingVerdict, TrainingDecision, NotifyState, EvaluationStatus, EvaluationItemStatus, EvaluationWinner, EvaluationItemWinnerFilter, EvaluationWarningCode, TrainingToolCall, TrainingMessage, TrainingRuleServing, TrainingRunPromoteRequest, TrainingRunRejectRequest, TrainingRunRollbackRequest, TrainingRunCancelRequest, TrainingRunListParams, TrainingRun, TrainingRunSummary, TrainingRunEvent, TrainingRunLinks, TrainingRunActions, EvaluationRef, EvaluationDecoding, EvaluationJudgeDimension, EvaluationDimensionScore, EvaluationWarning, EvaluationMargin, Evaluation, EvaluationItemSide, EvaluationItem, EvaluationItemListParams, BenchmarkStatus, BenchmarkSourceKind, BenchmarkRunStatus, BenchmarkScoring, BenchmarkDimensionScore, Benchmark, BenchmarkItem, BenchmarkRun, BenchmarkHistoryPoint, BenchmarkSource, BenchmarkCreateParams, BenchmarkListParams, BenchmarkItemListParams, BenchmarkHistoryParams, BenchmarkRetireParams, InferenceAlias, InferenceAliasRequest, TrainingRuleDeleteResponse, TrainingRunListResponse, TrainingRunResponse, TrainingRunActionResponse, EvaluationItemsResponse, InferenceAliasListResponse, InferenceAliasResponse, InferenceAliasDeleteResponse, BenchmarkListResponse, BenchmarkItemsResponse, BenchmarkHistoryResponse, TrainingRuleRunRefusalCode, TrainingRuleSaveRefusalCode, TrainingRuleMonthlyLimitRefusal, PipelineStatus, PipelineActiveRun, PipelineVersion, PipelineAttempt, PipelineAttemptTrigger, PipelineAttemptOutcome, Pipeline, PipelineListParams, PipelineListResponse, PipelineResponse, PipelineTrainWhen, PipelineSchedule, PipelineWriteRequest, PipelineMutationResponse, PipelineImportParams, PipelineFloors, LoopModel, LoopModelsResponse, LoopTracePipeline, LoopTracePipelinesResponse, LoopMetricsHistoryParams, LoopMetricPoint, LoopSignalDay, LoopMetricsHistory, PromotionPolicyState, RuleBenchmark, PipelineBenchmarksResponse, PipelineBenchmarkAttachParams, PipelineBenchmarkMoveParams, CompareBenchmark, VersionBenchmarkScore, PromotionMeasurement, PromotionOutcome, VersionScores, VersionScoreDifference, VersionCompareResponse, VersionCompareParams, } from './types.js';
78
+ export type { BiOSConfig, PaginatedResponse, ApiErrorBody, AvailableGpuAlternative, CapacityMinimumRequirement, InferenceBookingAccepted, Model, ModelSearchParams, ModelSearchResponse, ModelDetailResponse, InferenceModel, InferenceModelListResponse, ModelConfig, Dataset, DatasetListParams, DatasetListResponse, DatasetUploadParams, DatasetPreview, DatasetPreviewParams, DatasetImportHFParams, DatasetRegisterHFParams, DatasetHubSearchParams, DatasetHubPreviewParams, DatasetValidation, DatasetFormatVariant, DatasetFormatSpec, DatasetFormatSpecs, DatasetStorageUsage, TrainingMethod, RLHFAlgorithm, AdapterType, TrainingJobStatus, TrainingStatusFilter, TrainingCreateParams, TrainingListParams, TrainingListResponse, TrainingJob, TrainingMetrics, TrainingResourceMetrics, TrainingDeviceMetrics, MetricPoint, MetricGraphConfig, TrainingCheckpoint, TrainingLogs, TrainingLogEntry, TrainingStopResponse, TrainingResumeResponse, CanonicalTrainingRequest, TrainingPreflightDataset, TrainingPreflightWarning, TrainingPreflightResponse, TrainingCapabilityChoice, TrainingConfigFieldCapability, TrainingCapabilities, GPUChoice, WalletBalance, Transaction, TransactionListResponse, TransactionListParams, GPUInfo, GPUPricingResponse, GPUOptionsParams, TrainingRecommendParams, TrainingAdvisorVerdict, TrainingAdvisorRecommendation, TrainingAdvisorUserShape, TrainingAdvisorJustification, TrainingAdvisorBasis, GPUOption, GPUOptionSuggestion, GPUOptionsResponse, InferenceStatus, InferenceCreateParams, InferenceUpdateParams, InferenceDeployment, InferenceDeploymentSummary, InferenceCreateResponse, InferenceListResponse, InferenceUpdateResponse, InferenceLifecycleResponse, InferenceDeleteResponse, InferenceGPUOptionMarket, InferenceGPUOption, InferenceGPUAlternative, InferenceGPUOptionsParams, InferenceGPUOptionsResponse, InferenceCanonicalRequest, InferencePreflightWarning, InferencePreflightResponse, AdapterCompatibility, AdapterCompatibilityResponse, AdapterCompatibilityParams, SupportedArchitecture, ArchitectureScope, SupportedArchitecturesResponse, ApiKey, ApiKeyScope, ApiKeyIntrospection, Organization, OrgMember, OrgInvite, Workspace, Integration, IntegrationCreateParams, StorageUsage, StorageObject, PromptTokensDetails, ChatCompletionUsage, ServingKind, TrainingRulePausedReason, TrainingTrigger, TrainingRecipe, TrainingRunState, TrainingVerdict, TrainingDecision, EvaluationStatus, EvaluationItemStatus, EvaluationWinner, EvaluationItemWinnerFilter, EvaluationWarningCode, TrainingToolCall, TrainingMessage, TrainingRuleServing, TrainingRunPromoteRequest, TrainingRunRejectRequest, TrainingRunRollbackRequest, TrainingRunCancelRequest, TrainingRunListParams, TrainingRun, TrainingRunSummary, TrainingRunEvent, TrainingRunLinks, TrainingRunActions, EvaluationRef, EvaluationDecoding, EvaluationJudgeDimension, EvaluationDimensionScore, EvaluationWarning, EvaluationMargin, Evaluation, EvaluationItemSide, EvaluationItem, EvaluationItemListParams, BenchmarkStatus, BenchmarkSourceKind, BenchmarkRunStatus, BenchmarkScoring, BenchmarkDimensionScore, Benchmark, BenchmarkItem, BenchmarkRun, BenchmarkHistoryPoint, BenchmarkSource, BenchmarkCreateParams, BenchmarkListParams, BenchmarkItemListParams, BenchmarkHistoryParams, BenchmarkRetireParams, InferenceAlias, InferenceAliasRequest, TrainingRuleDeleteResponse, TrainingRunListResponse, TrainingRunResponse, TrainingRunActionResponse, EvaluationItemsResponse, InferenceAliasListResponse, InferenceAliasResponse, InferenceAliasDeleteResponse, BenchmarkListResponse, BenchmarkItemsResponse, BenchmarkHistoryResponse, TrainingRuleRunRefusalCode, TrainingRuleSaveRefusalCode, TrainingRuleMonthlyLimitRefusal, PipelineStatus, PipelineActiveRun, PipelineVersion, PipelineAttempt, PipelineAttemptTrigger, PipelineAttemptOutcome, Pipeline, PipelineListParams, PipelineListResponse, PipelineResponse, PipelineTrainWhen, PipelineSchedule, PipelineWriteRequest, PipelineMutationResponse, PipelineImportParams, PipelineFloors, LoopModel, LoopModelsResponse, LoopTracePipeline, LoopTracePipelinesResponse, LoopMetricsHistoryParams, LoopMetricPoint, LoopSignalDay, LoopMetricsHistory, PromotionPolicyState, RuleBenchmark, PipelineBenchmarksResponse, PipelineBenchmarkAttachParams, PipelineBenchmarkMoveParams, CompareBenchmark, VersionBenchmarkScore, PromotionMeasurement, PromotionOutcome, VersionScores, VersionScoreDifference, VersionCompareResponse, VersionCompareParams, } from './types.js';
79
79
  /** The closed set of run states a training run never leaves. */
80
80
  export { TERMINAL_RUN_STATES } from './types.js';
81
81
  /** The most versions a pipeline may be set to make. */
package/dist/index.js CHANGED
@@ -36,7 +36,7 @@ import { Loop } from './resources/loop.js';
36
36
  * SDK version. Sent as part of the User-Agent header.
37
37
  * Must match package.json "version" -- enforced by a contract test.
38
38
  */
39
- export const VERSION = '0.2.19-dev.277';
39
+ export const VERSION = '0.2.19-dev.281';
40
40
  export class RunBiOS {
41
41
  /** Search models, fetch configs, check adapter compatibility. */
42
42
  models;
@@ -402,7 +402,9 @@ export declare class Loop {
402
402
  * numbers `minimums` shows. 503 `AGENT_COULD_NOT_START` means the key the
403
403
  * attempt would be scored with could not be renewed just then: nothing was
404
404
  * started, so press train now again in a minute. Every refusal is a
405
- * `TrainingRuleRunRefusalCode`.
405
+ * `TrainingRuleRunRefusalCode`. When the attempt in flight is waiting for
406
+ * GPUs to compare (`active_run.state` `waiting_gpus`), this books its two
407
+ * comparison machines again at once instead of starting another attempt.
406
408
  */
407
409
  runPipeline(id: string): Promise<PipelineMutationResponse>;
408
410
  /**
@@ -555,7 +555,9 @@ export class Loop {
555
555
  * numbers `minimums` shows. 503 `AGENT_COULD_NOT_START` means the key the
556
556
  * attempt would be scored with could not be renewed just then: nothing was
557
557
  * started, so press train now again in a minute. Every refusal is a
558
- * `TrainingRuleRunRefusalCode`.
558
+ * `TrainingRuleRunRefusalCode`. When the attempt in flight is waiting for
559
+ * GPUs to compare (`active_run.state` `waiting_gpus`), this books its two
560
+ * comparison machines again at once instead of starting another attempt.
559
561
  */
560
562
  async runPipeline(id) {
561
563
  return this._http.fetchPost(`/api/loop/pipelines/${encodeURIComponent(id)}/run`, {});
package/dist/types.d.ts CHANGED
@@ -2611,6 +2611,11 @@ export interface LoopTrace {
2611
2611
  verdict?: LoopVerdict;
2612
2612
  /** Who gave that verdict. */
2613
2613
  verdict_source?: LoopSource;
2614
+ /**
2615
+ * The score a `scored` verdict carries: above zero it trains the answer as
2616
+ * it stands (good feedback), zero or below it trains nothing (bad).
2617
+ */
2618
+ verdict_score?: number;
2614
2619
  /** Its tags. On the single read only. */
2615
2620
  labels?: LoopLabel[];
2616
2621
  /**
@@ -2861,12 +2866,22 @@ export interface TrainingRecipe {
2861
2866
  lora_rank: number | null;
2862
2867
  lora_alpha: number | null;
2863
2868
  }
2864
- export type TrainingRunState = 'built' | 'registering' | 'dataset_validating' | 'preflighting' | 'job_creating' | 'training' | 'checkpoint_ready' | 'candidate_booking' | 'candidate_running' | 'evaluating' | 'reported' | 'awaiting_review' | 'promoting' | 'waiting_funds' | 'promoted' | 'rejected' | 'expired' | 'failed' | 'budget_stopped' | 'cancelled' | 'superseded' | 'rolled_back';
2869
+ export type TrainingRunState = 'built' | 'registering' | 'dataset_validating' | 'preflighting' | 'job_creating' | 'training' | 'checkpoint_ready' | 'candidate_booking' | 'candidate_running' | 'evaluating' | 'reported' | 'awaiting_review' | 'promoting' | 'waiting_funds'
2870
+ /**
2871
+ * Trained, and waiting for GPUs to run the new model and the one it is
2872
+ * compared with side by side: one machine got a GPU and the other did not
2873
+ * within 20 minutes, so both were stopped. Each try can bill the machine
2874
+ * that got a GPU for up to those 20 minutes; nothing bills between tries.
2875
+ * It keeps what it trained, books both again after 30 minutes, then 1, 2
2876
+ * and 4 hours, and ends (`failed`, `COMPARISON_NO_GPUS`) after four tries or
2877
+ * 24 hours (`COMPARISON_WORKSPACE_FULL` when what was missing was room in
2878
+ * the workspace). Train now books them again at once.
2879
+ */
2880
+ | 'waiting_gpus' | 'promoted' | 'rejected' | 'expired' | 'failed' | 'budget_stopped' | 'cancelled' | 'superseded' | 'rolled_back';
2865
2881
  /** The closed set a run never leaves. */
2866
2882
  export declare const TERMINAL_RUN_STATES: readonly TrainingRunState[];
2867
2883
  export type TrainingVerdict = 'better' | 'not_better' | 'inconclusive' | 'not_evaluated';
2868
2884
  export type TrainingDecision = 'auto_promoted' | 'promoted' | 'rejected' | 'auto_rejected' | 'rolled_back';
2869
- export type NotifyState = 'none' | 'pending' | 'sending' | 'sent' | 'suppressed' | 'failed';
2870
2885
  export type EvaluationStatus = 'open' | 'done' | 'failed' | 'budget_stopped';
2871
2886
  export type EvaluationItemStatus = 'pending' | 'working' | 'done' | 'failed';
2872
2887
  export type EvaluationWinner = 'candidate' | 'incumbent' | 'tie';
@@ -2953,7 +2968,19 @@ export interface TrainingRun {
2953
2968
  */
2954
2969
  error_next_step: string | null;
2955
2970
  billed_training_cents: number;
2971
+ /**
2972
+ * The comparison machines: each from the first time it was on a GPU to its
2973
+ * stop, less any time back in the GPU queue, at the price the platform
2974
+ * placed it at.
2975
+ */
2956
2976
  billed_candidate_cents: number;
2977
+ /**
2978
+ * True when a comparison machine was charged: the figure is then the most
2979
+ * it can have cost ("up to"), since it counts each machine from its first
2980
+ * boot where the platform bills from its first answer (and at the hourly
2981
+ * cap when no placement price was reported).
2982
+ */
2983
+ billed_candidate_up_to: boolean;
2957
2984
  spent_eval_cents: number;
2958
2985
  train_rows: number | null;
2959
2986
  holdout_rows: number | null;
@@ -3008,10 +3035,6 @@ export interface TrainingRun {
3008
3035
  decided_at: string | null;
3009
3036
  auto_promote_at: string | null;
3010
3037
  review_deadline_at: string | null;
3011
- notify_state: NotifyState;
3012
- notify_event: string | null;
3013
- notify_error: string | null;
3014
- notify_attempts: number;
3015
3038
  created_at: string;
3016
3039
  updated_at: string;
3017
3040
  finished_at: string | null;
@@ -3049,6 +3072,8 @@ export interface TrainingRunSummary {
3049
3072
  holdout_rows: number | null;
3050
3073
  billed_training_cents: number;
3051
3074
  billed_candidate_cents: number;
3075
+ /** See {@link TrainingRun.billed_candidate_up_to}. */
3076
+ billed_candidate_up_to: boolean;
3052
3077
  spent_eval_cents: number;
3053
3078
  /** The fourth money column. A row that leaves it out adds up short. */
3054
3079
  benchmark_spent_cents: number;
@@ -3057,7 +3082,6 @@ export interface TrainingRunSummary {
3057
3082
  evaluation_id: string | null;
3058
3083
  review_deadline_at: string | null;
3059
3084
  auto_promote_at: string | null;
3060
- notify_state: NotifyState;
3061
3085
  created_at: string;
3062
3086
  finished_at: string | null;
3063
3087
  }
@@ -3515,8 +3539,8 @@ export type TrainingRuleSaveRefusalCode = 'PIPELINE_NAME_TAKEN' | 'OWN_TAG_TAKEN
3515
3539
  *
3516
3540
  * The test is the worst case, not the average: a run may start only if the
3517
3541
  * most the rule's runs this UTC calendar month can have cost
3518
- * (`month_spent_cents`), plus the most the run about to start may spend (its
3519
- * three ceilings added up), fits under the limit.
3542
+ * (`month_spent_cents`), plus the most the run about to start may spend
3543
+ * (`run_max_cents`), fits under the limit.
3520
3544
  */
3521
3545
  export interface TrainingRuleMonthlyLimitRefusal {
3522
3546
  error: {
@@ -3527,7 +3551,11 @@ export interface TrainingRuleMonthlyLimitRefusal {
3527
3551
  month_spent_cents: number;
3528
3552
  /** The rule's monthly limit. */
3529
3553
  monthly_ceiling_cents: number;
3530
- /** The most one run of this rule may spend: training + candidate + evaluation ceilings. */
3554
+ /**
3555
+ * The most one attempt of this rule may spend: its training, its comparison
3556
+ * machine and its judging, and -- on a pipeline, while no version is live --
3557
+ * the base model's own comparison machine beside the new model's.
3558
+ */
3531
3559
  run_max_cents: number;
3532
3560
  /**
3533
3561
  * The first instant of the next UTC month, when the counted spend starts
@@ -3571,6 +3599,11 @@ export interface PipelineActiveRun {
3571
3599
  gpu_seconds: number | null;
3572
3600
  /** What it has been billed so far, in cents. */
3573
3601
  cost_cents: number;
3602
+ /**
3603
+ * True when a comparison machine was charged in `cost_cents`: the figure is
3604
+ * the most it can have cost, so show it as "up to".
3605
+ */
3606
+ cost_up_to: boolean;
3574
3607
  /** The version number it will take if it is put live. */
3575
3608
  will_be_version: number;
3576
3609
  /** Always null: nothing in flight is a version yet. */
@@ -3643,6 +3676,14 @@ export interface PipelineVersion {
3643
3676
  * not what it did.
3644
3677
  */
3645
3678
  cost_includes_candidate_ceiling: boolean;
3679
+ /**
3680
+ * True when `cost_cents` is the most the version can have cost rather than
3681
+ * what it did: its comparison machines at their ceilings
3682
+ * (`cost_includes_candidate_ceiling`), or one charged (counted from its
3683
+ * first boot, where the platform bills from its first answer). Show it as
3684
+ * "up to".
3685
+ */
3686
+ cost_up_to: boolean;
3646
3687
  /** Machine time the attempt held (training and comparison); null when unknown. */
3647
3688
  gpu_seconds: number | null;
3648
3689
  created_at: string;
@@ -3691,6 +3732,8 @@ export interface PipelineAttempt {
3691
3732
  gpu_seconds: number | null;
3692
3733
  /** As the versions are priced once it has ended; as recorded so far while it runs. */
3693
3734
  cost_cents: number;
3735
+ /** See {@link PipelineVersion.cost_up_to}. */
3736
+ cost_up_to: boolean;
3694
3737
  verdict: TrainingVerdict | null;
3695
3738
  win_rate: number | null;
3696
3739
  evaluation_id: string | null;
@@ -3820,10 +3863,11 @@ export interface Pipeline {
3820
3863
  * this UTC calendar month -- the figure the limit is enforced against. An
3821
3864
  * upper bound, not an exact spend. A run counts toward the month it was
3822
3865
  * created in, and a run still in progress also counts toward the current
3823
- * month, at its three ceilings or what it has been billed when that is
3824
- * more. A finished run counts what it was billed, and its comparison
3825
- * machine is billed at the most it could have cost (the hourly cap for the
3826
- * time it was up, never more than the candidate ceiling). A run created in
3866
+ * month, at the most it may cost (as `run_max_cents` counts it, the base
3867
+ * model's comparison machine included) or what it has been billed when that
3868
+ * is more. A finished run counts what it was billed, and its comparison
3869
+ * machines are billed at the most they could have cost (each machine's
3870
+ * hourly cap for the time it was up, never more than its own amount). A run created in
3827
3871
  * an earlier month that finishes in this one counts here only while it is
3828
3872
  * still running.
3829
3873
  */
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "runbios-sdk",
3
- "version": "0.2.19-dev.277",
3
+ "version": "0.2.19-dev.281",
4
4
  "description": "Official TypeScript SDK for the Run BiOS training and deployment platform API",
5
5
  "type": "module",
6
6
  "main": "./dist/index.js",