runbios-sdk 0.2.14-dev.246 → 0.2.14-dev.248

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/README.md CHANGED
@@ -186,6 +186,37 @@ if (answer) {
186
186
  Streaming billing is charged server-side on completed usage; the SDK only needs
187
187
  to request `usage` where the endpoint exposes it (no client change).
188
188
 
189
+ #### Other supported `/v1` inference tasks
190
+
191
+ Serverless catalogs serve OpenAI chat and, when the model supports that dialect,
192
+ Anthropic Messages (`inference.messages` / `streamMessages`). Serverless does
193
+ **not** serve `/v1/completions`, `/v1/embeddings`, or `/v1/rerank`. Those three
194
+ routes require a dedicated deployment that actually advertises the matching
195
+ task and a credential with `deployments:read` or `deployments:write`; a chat-only
196
+ deployment cannot embed or rerank. The server's preflight/status result, not the
197
+ model name, determines serving mode. The SDK forwards a dedicated `inferenceKey`
198
+ when configured, otherwise the workspace platform `apiKey`.
199
+
200
+ ```typescript
201
+ import { Inference } from 'runbios-sdk';
202
+ const reply = await client.inference.messages({
203
+ model: 'catalog-model-with-messages-support', max_tokens: 64,
204
+ messages: [{ role: 'user', content: 'Hello.' }],
205
+ });
206
+
207
+ const dedicated = new Inference();
208
+ const text = await dedicated.completions({ model: 'completion-deployment', prompt: 'Continue' });
209
+ const vectors = await dedicated.embeddings({ model: 'embedding-deployment', input: ['First', 'Second'] });
210
+ const ranking = await dedicated.rerank({ model: 'rerank-deployment', query: 'Question', documents: ['A', 'B'] });
211
+ for await (const event of dedicated.streamCompletions({ model: 'completion-deployment', prompt: 'Continue' })) {
212
+ console.log(event);
213
+ }
214
+ ```
215
+
216
+ `streamMessages` also yields each Anthropic SSE event in order. Both streaming
217
+ methods close the upstream reader when iteration ends. None of the inference
218
+ POSTs are retried automatically, and `idempotencyKey` is sent only if supplied.
219
+
189
220
  #### Workspace serverless usage and limits (read-only)
190
221
 
191
222
  A workspace-bound platform key or hosted OAuth grant with `analytics:read` can
package/dist/index.d.ts CHANGED
@@ -36,7 +36,7 @@ import { Loop } from './resources/loop.js';
36
36
  * SDK version. Sent as part of the User-Agent header.
37
37
  * Must match package.json "version" -- enforced by a contract test.
38
38
  */
39
- export declare const VERSION = "0.2.14-dev.246";
39
+ export declare const VERSION = "0.2.14-dev.248";
40
40
  export declare class RunBiOS {
41
41
  /** Search models, fetch configs, check adapter compatibility. */
42
42
  readonly models: Models;
@@ -74,7 +74,7 @@ export { Integrations, type HuggingFaceIntegration, type IntegrationBrowseParams
74
74
  export { Training } from './resources/training.js';
75
75
  export { Wallet } from './resources/wallet.js';
76
76
  export { GPU, type GPURecommendation } from './resources/gpu.js';
77
- export { Inference, validateChatRequest, parseSSE, buildInferenceRequest, contextSizingBasis, type ContextSizingBasis, CONTEXT_DEFAULT_CEILING, CONTEXT_EDITABLE_FLOOR, type ChatMessage, type FunctionTool, type ChatCompletionParams, type ChatCompletionResponse, type ChatCompletionChunk, type ServerlessUsageWindow, type ServerlessTimeseriesMetric, type ServerlessWorkspaceLimits, type ServerlessUsageEnvelope, type ServerlessUsageOverviewResponse, type ServerlessUsageRowsResponse, type ServerlessUsageTimeseriesResponse, type ServerlessUsageDailyResponse, type ServerlessSavingsTotals, type ServerlessSavingsPeriod, type ServerlessUsageSavingsResponse, } from './resources/inference.js';
77
+ export { Inference, validateChatRequest, parseSSE, buildInferenceRequest, contextSizingBasis, type ContextSizingBasis, CONTEXT_DEFAULT_CEILING, CONTEXT_EDITABLE_FLOOR, type ChatMessage, type FunctionTool, type ChatCompletionParams, type ChatCompletionResponse, type ChatCompletionChunk, type InferenceVerbParams, type CompletionParams, type EmbeddingParams, type RerankParams, type AnthropicMessageParams, type ServerlessUsageWindow, type ServerlessTimeseriesMetric, type ServerlessWorkspaceLimits, type ServerlessUsageEnvelope, type ServerlessUsageOverviewResponse, type ServerlessUsageRowsResponse, type ServerlessUsageTimeseriesResponse, type ServerlessUsageDailyResponse, type ServerlessSavingsTotals, type ServerlessSavingsPeriod, type ServerlessUsageSavingsResponse, } from './resources/inference.js';
78
78
  export type { BiOSConfig, PaginatedResponse, ApiErrorBody, AvailableGpuAlternative, CapacityMinimumRequirement, InferenceBookingAccepted, Model, ModelSearchParams, ModelSearchResponse, ModelDetailResponse, InferenceModel, InferenceModelListResponse, ModelConfig, Dataset, DatasetListParams, DatasetListResponse, DatasetUploadParams, DatasetPreview, DatasetPreviewParams, DatasetImportHFParams, DatasetRegisterHFParams, DatasetHubSearchParams, DatasetHubPreviewParams, DatasetValidation, DatasetFormatVariant, DatasetFormatSpec, DatasetFormatSpecs, DatasetStorageUsage, TrainingMethod, RLHFAlgorithm, AdapterType, TrainingJobStatus, TrainingStatusFilter, TrainingCreateParams, TrainingListParams, TrainingListResponse, TrainingJob, TrainingMetrics, TrainingResourceMetrics, TrainingDeviceMetrics, MetricPoint, MetricGraphConfig, TrainingCheckpoint, TrainingLogs, TrainingLogEntry, TrainingStopResponse, TrainingResumeResponse, CanonicalTrainingRequest, TrainingPreflightDataset, TrainingPreflightWarning, TrainingPreflightResponse, TrainingCapabilityChoice, TrainingConfigFieldCapability, TrainingCapabilities, GPUChoice, WalletBalance, Transaction, TransactionListResponse, TransactionListParams, GPUInfo, GPUPricingResponse, GPUOptionsParams, TrainingRecommendParams, TrainingAdvisorVerdict, TrainingAdvisorRecommendation, TrainingAdvisorUserShape, TrainingAdvisorJustification, TrainingAdvisorBasis, GPUOption, GPUOptionSuggestion, GPUOptionsResponse, InferenceStatus, InferenceCreateParams, InferenceUpdateParams, InferenceDeployment, InferenceDeploymentSummary, InferenceCreateResponse, InferenceListResponse, InferenceUpdateResponse, InferenceLifecycleResponse, InferenceDeleteResponse, InferenceGPUOptionMarket, InferenceGPUOption, InferenceGPUAlternative, InferenceGPUOptionsParams, InferenceGPUOptionsResponse, InferenceCanonicalRequest, InferencePreflightWarning, InferencePreflightResponse, AdapterCompatibility, AdapterCompatibilityResponse, AdapterCompatibilityParams, SupportedArchitecture, ArchitectureScope, SupportedArchitecturesResponse, ApiKey, ApiKeyScope, ApiKeyIntrospection, Organization, OrgMember, OrgInvite, Workspace, Integration, IntegrationCreateParams, StorageUsage, StorageObject, PromptTokensDetails, ChatCompletionUsage, TrainingCadence, TrainingCombinator, ServingKind, TrainingRulePausedReason, LoopTrainingMethod, TrainType, GradersScope, TrainingTrigger, TrainingRecipe, TrainingRunState, TrainingVerdict, TrainingDecision, NotifyState, EvaluationStatus, EvaluationItemStatus, EvaluationWinner, EvaluationWarningCode, ConsentVia, TrainingToolCall, TrainingMessage, TrainingRuleServing, TrainingGPURung, TrainingRule, TrainingRuleConsent, TrainingRulePreflightRefusal, TrainingRuleKeyRef, TrainingRuleKeyCheck, TrainingRulePreflight, TrainingRuleBuildSpec, TrainingRuleTriggerInput, TrainingRuleTrainingInput, TrainingRuleDeployInput, TrainingRuleMoneyInput, TrainingRuleEvaluationInput, TrainingRulePromotionInput, TrainingRuleAcceptTerms, TrainingRulePreflightRequest, TrainingRuleCreateRequest, TrainingRuleUpdateRequest, TrainingRuleConsentRequest, TrainingRuleListParams, TrainingRunPromoteRequest, TrainingRunRejectRequest, TrainingRunRollbackRequest, TrainingRunCancelRequest, TrainingRunListParams, TrainingRun, TrainingRunSummary, TrainingRunEvent, TrainingRunLinks, TrainingRunActions, EvaluationRef, EvaluationDecoding, EvaluationJudgeDimension, EvaluationDimensionScore, EvaluationGraderScore, EvaluationWarning, EvaluationMargin, JudgeAgreementDimension, JudgeAgreement, Evaluation, EvaluationItemSide, EvaluationItem, EvaluationItemListParams, JudgeAgreementParams, AgentSettings, AgentSettingsRequest, BenchmarkStatus, BenchmarkSourceKind, BenchmarkRunStatus, BenchmarkScoring, BenchmarkDimensionScore, Benchmark, BenchmarkItem, BenchmarkRun, BenchmarkHistoryPoint, BenchmarkSource, BenchmarkCreateParams, BenchmarkListParams, BenchmarkItemListParams, BenchmarkHistoryParams, BenchmarkRetireParams, TrainingRuleBenchmarkRequest, InferenceAlias, InferenceAliasRequest, TrainingRuleListResponse, TrainingRuleResponse, TrainingRuleMutationResponse, TrainingRuleDeleteResponse, TrainingRunListResponse, TrainingRunResponse, TrainingRunActionResponse, EvaluationItemsResponse, AgentSettingsResponse, InferenceAliasListResponse, InferenceAliasResponse, InferenceAliasDeleteResponse, BenchmarkListResponse, BenchmarkItemsResponse, BenchmarkHistoryResponse, TrainingRuleRunRefusalCode, TrainingRuleMonthlyLimitRefusal, LoopDatasetDeleteRefusalCode, PipelineStatus, PipelineRecipeExploration, PipelineActiveRun, PipelineVersion, Pipeline, PipelineListResponse, PipelineResponse, } from './types.js';
79
79
  /** The closed set of run states a training run never leaves. */
80
80
  export { TERMINAL_RUN_STATES } from './types.js';
package/dist/index.js CHANGED
@@ -36,7 +36,7 @@ import { Loop } from './resources/loop.js';
36
36
  * SDK version. Sent as part of the User-Agent header.
37
37
  * Must match package.json "version" -- enforced by a contract test.
38
38
  */
39
- export const VERSION = '0.2.14-dev.246';
39
+ export const VERSION = '0.2.14-dev.248';
40
40
  export class RunBiOS {
41
41
  /** Search models, fetch configs, check adapter compatibility. */
42
42
  models;
@@ -78,6 +78,28 @@ export interface ChatCompletionParams extends Record<string, unknown> {
78
78
  }
79
79
  export type ChatCompletionResponse = Record<string, unknown>;
80
80
  export type ChatCompletionChunk = Record<string, unknown>;
81
+ export interface InferenceVerbParams extends Record<string, unknown> {
82
+ model?: string;
83
+ inferenceKey?: string;
84
+ idempotencyKey?: string;
85
+ requestId?: string;
86
+ signal?: AbortSignal;
87
+ }
88
+ export interface CompletionParams extends InferenceVerbParams {
89
+ prompt: string | string[] | number[] | number[][];
90
+ }
91
+ export interface EmbeddingParams extends InferenceVerbParams {
92
+ input: string | string[] | number[] | number[][];
93
+ }
94
+ export interface RerankParams extends InferenceVerbParams {
95
+ query: string;
96
+ documents: Array<string | Record<string, unknown>>;
97
+ }
98
+ export interface AnthropicMessageParams extends InferenceVerbParams {
99
+ messages: Array<Record<string, unknown>>;
100
+ max_tokens: number;
101
+ anthropicVersion?: string;
102
+ }
81
103
  export type ServerlessUsageWindow = '1h' | '24h' | '7d' | '30d' | '90d';
82
104
  export type ServerlessTimeseriesMetric = 'requests' | 'tokens' | 'spend' | 'ttft_p50' | 'ttft_p95' | 'tps';
83
105
  export interface ServerlessWorkspaceLimits {
@@ -151,8 +173,8 @@ export declare function validateChatRequest(body: Record<string, unknown>): void
151
173
  export declare function parseSSE(body: ReadableStream<Uint8Array>): AsyncGenerator<string>;
152
174
  /**
153
175
  * Inference surface. Combines control-plane management of model-serving
154
- * deployments (`/api/inference*`) with OpenAI-compatible key-scoped inference
155
- * (`/v1/chat/completions`). Requests are dispatched once; an idempotency header
176
+ * deployments (`/api/inference*`) with key-scoped inference on the supported
177
+ * `/v1` task routes. Requests are dispatched once; an idempotency header
156
178
  * is forwarded but server-side replay is not assumed.
157
179
  */
158
180
  export declare class Inference {
@@ -330,7 +352,16 @@ export declare class Inference {
330
352
  getGPUOptions(params: InferenceGPUOptionsParams): Promise<InferenceGPUOptionsResponse>;
331
353
  private prepare;
332
354
  private abortContext;
355
+ private prepareVerb;
356
+ private sendVerb;
333
357
  private modelRead;
358
+ private streamVerb;
359
+ completions(params: CompletionParams): Promise<Record<string, unknown>>;
360
+ embeddings(params: EmbeddingParams): Promise<Record<string, unknown>>;
361
+ rerank(params: RerankParams): Promise<Record<string, unknown>>;
362
+ messages(params: AnthropicMessageParams): Promise<Record<string, unknown>>;
363
+ streamCompletions(params: CompletionParams): AsyncGenerator<Record<string, unknown>>;
364
+ streamMessages(params: AnthropicMessageParams): AsyncGenerator<Record<string, unknown>>;
334
365
  listModels(inferenceKey?: string): Promise<InferenceModelListResponse>;
335
366
  retrieveModel(modelId: string, inferenceKey?: string): Promise<InferenceModel>;
336
367
  chatCompletions(params: ChatCompletionParams): Promise<ChatCompletionResponse>;
@@ -318,8 +318,8 @@ async function apiError(response) {
318
318
  }
319
319
  /**
320
320
  * Inference surface. Combines control-plane management of model-serving
321
- * deployments (`/api/inference*`) with OpenAI-compatible key-scoped inference
322
- * (`/v1/chat/completions`). Requests are dispatched once; an idempotency header
321
+ * deployments (`/api/inference*`) with key-scoped inference on the supported
322
+ * `/v1` task routes. Requests are dispatched once; an idempotency header
323
323
  * is forwarded but server-side replay is not assumed.
324
324
  */
325
325
  export class Inference {
@@ -830,6 +830,61 @@ export class Inference {
830
830
  remove: () => signal?.removeEventListener('abort', relay),
831
831
  };
832
832
  }
833
+ prepareVerb(params, stream, anthropic) {
834
+ const { inferenceKey, idempotencyKey, requestId, signal, anthropicVersion, stream: requestedStream, ...payload } = params;
835
+ if (requestedStream !== undefined)
836
+ throw new Error('Choose the streaming method instead of setting stream in a request body');
837
+ const key = inferenceKey || this.key;
838
+ if (!key)
839
+ throw new Error('an inferenceKey is required');
840
+ const headers = {
841
+ Authorization: `Bearer ${key}`,
842
+ Accept: stream ? 'text/event-stream' : 'application/json',
843
+ 'Content-Type': 'application/json',
844
+ 'X-Request-ID': requestId || crypto.randomUUID(),
845
+ };
846
+ if (idempotencyKey)
847
+ headers['Idempotency-Key'] = idempotencyKey;
848
+ if (anthropic) {
849
+ if (anthropicVersion !== undefined && typeof anthropicVersion !== 'string') {
850
+ throw new Error('anthropicVersion must be a string');
851
+ }
852
+ headers['Anthropic-Version'] = anthropicVersion || '2023-06-01';
853
+ }
854
+ return { body: stream ? { ...payload, stream: true } : payload, headers, signal };
855
+ }
856
+ async sendVerb(path, params, anthropic = false) {
857
+ const prepared = this.prepareVerb(params, false, anthropic);
858
+ const abort = this.abortContext(prepared.signal);
859
+ try {
860
+ const response = await fetch(`${this.baseUrl}${path}`, {
861
+ method: 'POST', headers: prepared.headers, body: JSON.stringify(prepared.body), signal: abort.controller.signal,
862
+ });
863
+ if (!response.ok)
864
+ throw await apiError(response);
865
+ let payload;
866
+ try {
867
+ payload = await response.json();
868
+ }
869
+ catch {
870
+ throw new ApiError(response.status, { error: 'Inference endpoint returned invalid JSON' });
871
+ }
872
+ if (!payload || typeof payload !== 'object' || Array.isArray(payload)) {
873
+ throw new ApiError(response.status, { error: 'Inference endpoint returned no response object' });
874
+ }
875
+ return payload;
876
+ }
877
+ catch (error) {
878
+ if (abort.controller.signal.aborted && !(error instanceof ApiError)) {
879
+ throw new ApiError(0, { error: String(abort.controller.signal.reason || 'Inference request aborted') });
880
+ }
881
+ throw error;
882
+ }
883
+ finally {
884
+ clearTimeout(abort.timeoutId);
885
+ abort.remove();
886
+ }
887
+ }
833
888
  async modelRead(path, inferenceKey) {
834
889
  const key = inferenceKey || this.key;
835
890
  if (!key)
@@ -861,6 +916,75 @@ export class Inference {
861
916
  abort.remove();
862
917
  }
863
918
  }
919
+ async *streamVerb(path, params, anthropic = false) {
920
+ const prepared = this.prepareVerb(params, true, anthropic);
921
+ const abort = this.abortContext(prepared.signal);
922
+ try {
923
+ const response = await fetch(`${this.baseUrl}${path}`, {
924
+ method: 'POST', headers: prepared.headers, body: JSON.stringify(prepared.body), signal: abort.controller.signal,
925
+ });
926
+ if (!response.ok)
927
+ throw await apiError(response);
928
+ const contentType = response.headers.get('content-type')?.toLowerCase() ?? '';
929
+ if (!contentType.includes('text/event-stream')) {
930
+ throw new ApiError(response.status, { error: `Expected text/event-stream, received ${contentType || 'no content type'}` });
931
+ }
932
+ if (!response.body)
933
+ throw new ApiError(0, { error: 'Inference stream returned no body' });
934
+ for await (const data of parseSSE(response.body)) {
935
+ if (data === '[DONE]') {
936
+ if (anthropic)
937
+ throw new ApiError(0, { error: 'Anthropic stream ended without message_stop' });
938
+ return;
939
+ }
940
+ let event;
941
+ try {
942
+ event = JSON.parse(data);
943
+ }
944
+ catch {
945
+ throw new ApiError(0, { error: 'Invalid JSON SSE event' });
946
+ }
947
+ if (!event || typeof event !== 'object' || Array.isArray(event)) {
948
+ throw new ApiError(0, { error: 'Inference stream returned an invalid event' });
949
+ }
950
+ if ('error' in event)
951
+ throw new ApiError(0, event);
952
+ yield event;
953
+ if (anthropic && event.type === 'message_stop')
954
+ return;
955
+ }
956
+ throw new ApiError(0, { error: 'Inference stream ended before completion marker' });
957
+ }
958
+ catch (error) {
959
+ if (abort.controller.signal.aborted && !(error instanceof ApiError)) {
960
+ throw new ApiError(0, { error: String(abort.controller.signal.reason || 'Inference request aborted') });
961
+ }
962
+ throw error;
963
+ }
964
+ finally {
965
+ abort.controller.abort('stream closed');
966
+ clearTimeout(abort.timeoutId);
967
+ abort.remove();
968
+ }
969
+ }
970
+ completions(params) {
971
+ return this.sendVerb('/v1/completions', params);
972
+ }
973
+ embeddings(params) {
974
+ return this.sendVerb('/v1/embeddings', params);
975
+ }
976
+ rerank(params) {
977
+ return this.sendVerb('/v1/rerank', params);
978
+ }
979
+ messages(params) {
980
+ return this.sendVerb('/v1/messages', params, true);
981
+ }
982
+ async *streamCompletions(params) {
983
+ yield* this.streamVerb('/v1/completions', params);
984
+ }
985
+ async *streamMessages(params) {
986
+ yield* this.streamVerb('/v1/messages', params, true);
987
+ }
864
988
  async listModels(inferenceKey) {
865
989
  return asInferenceModelList(await this.modelRead('/v1/models', inferenceKey));
866
990
  }
@@ -922,6 +1046,7 @@ export class Inference {
922
1046
  }
923
1047
  yield event;
924
1048
  }
1049
+ throw new ApiError(0, { error: 'Inference stream ended before completion marker [DONE]' });
925
1050
  }
926
1051
  catch (error) {
927
1052
  if (abort.controller.signal.aborted && !(error instanceof ApiError)) {
package/dist/types.d.ts CHANGED
@@ -1453,11 +1453,13 @@ export interface GPUOptionsResponse {
1453
1453
  availability_known: boolean;
1454
1454
  no_fit?: boolean;
1455
1455
  adapter_unavailable?: boolean;
1456
+ method_unavailable?: boolean;
1456
1457
  host_memory_unverified?: boolean;
1457
1458
  model_support?: {
1458
1459
  tier?: string;
1459
1460
  adapters?: string[];
1460
1461
  verified_adapters?: string[];
1462
+ admissible_method_adapters?: Record<string, string[]>;
1461
1463
  objectives?: string[];
1462
1464
  [key: string]: unknown;
1463
1465
  } | null;
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "runbios-sdk",
3
- "version": "0.2.14-dev.246",
3
+ "version": "0.2.14-dev.248",
4
4
  "description": "Official TypeScript SDK for the Run BiOS training and deployment platform API",
5
5
  "type": "module",
6
6
  "main": "./dist/index.js",