runbios-sdk 0.2.14-dev.245 → 0.2.14-dev.247
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +59 -0
- package/dist/index.d.ts +3 -3
- package/dist/index.js +1 -1
- package/dist/resources/inference.d.ts +37 -3
- package/dist/resources/inference.js +168 -3
- package/dist/resources/models.d.ts +5 -1
- package/dist/resources/models.js +26 -0
- package/dist/types.d.ts +16 -0
- package/package.json +1 -1
package/README.md
CHANGED
|
@@ -186,6 +186,37 @@ if (answer) {
|
|
|
186
186
|
Streaming billing is charged server-side on completed usage; the SDK only needs
|
|
187
187
|
to request `usage` where the endpoint exposes it (no client change).
|
|
188
188
|
|
|
189
|
+
#### Other supported `/v1` inference tasks
|
|
190
|
+
|
|
191
|
+
Serverless catalogs serve OpenAI chat and, when the model supports that dialect,
|
|
192
|
+
Anthropic Messages (`inference.messages` / `streamMessages`). Serverless does
|
|
193
|
+
**not** serve `/v1/completions`, `/v1/embeddings`, or `/v1/rerank`. Those three
|
|
194
|
+
routes require a dedicated deployment that actually advertises the matching
|
|
195
|
+
task and a credential with `deployments:read` or `deployments:write`; a chat-only
|
|
196
|
+
deployment cannot embed or rerank. The server's preflight/status result, not the
|
|
197
|
+
model name, determines serving mode. The SDK forwards a dedicated `inferenceKey`
|
|
198
|
+
when configured, otherwise the workspace platform `apiKey`.
|
|
199
|
+
|
|
200
|
+
```typescript
|
|
201
|
+
import { Inference } from 'runbios-sdk';
|
|
202
|
+
const reply = await client.inference.messages({
|
|
203
|
+
model: 'catalog-model-with-messages-support', max_tokens: 64,
|
|
204
|
+
messages: [{ role: 'user', content: 'Hello.' }],
|
|
205
|
+
});
|
|
206
|
+
|
|
207
|
+
const dedicated = new Inference();
|
|
208
|
+
const text = await dedicated.completions({ model: 'completion-deployment', prompt: 'Continue' });
|
|
209
|
+
const vectors = await dedicated.embeddings({ model: 'embedding-deployment', input: ['First', 'Second'] });
|
|
210
|
+
const ranking = await dedicated.rerank({ model: 'rerank-deployment', query: 'Question', documents: ['A', 'B'] });
|
|
211
|
+
for await (const event of dedicated.streamCompletions({ model: 'completion-deployment', prompt: 'Continue' })) {
|
|
212
|
+
console.log(event);
|
|
213
|
+
}
|
|
214
|
+
```
|
|
215
|
+
|
|
216
|
+
`streamMessages` also yields each Anthropic SSE event in order. Both streaming
|
|
217
|
+
methods close the upstream reader when iteration ends. None of the inference
|
|
218
|
+
POSTs are retried automatically, and `idempotencyKey` is sent only if supplied.
|
|
219
|
+
|
|
189
220
|
#### Workspace serverless usage and limits (read-only)
|
|
190
221
|
|
|
191
222
|
A workspace-bound platform key or hosted OAuth grant with `analytics:read` can
|
|
@@ -251,6 +282,34 @@ const compat = await client.models.getAdapterCompatibility({
|
|
|
251
282
|
});
|
|
252
283
|
```
|
|
253
284
|
|
|
285
|
+
#### Models this credential can invoke
|
|
286
|
+
|
|
287
|
+
`client.models.list()` reads the unified `GET /v1/models` roster with the SDK's
|
|
288
|
+
platform API key: serverless pool models and workspace deployments appear only
|
|
289
|
+
when that key's scopes permit them. `client.models.retrieve(id)` uses the same
|
|
290
|
+
scope and workspace boundary. The trainable `models.search()` / `models.get()`
|
|
291
|
+
registry above is separate; a registry result does not guarantee that this
|
|
292
|
+
credential can invoke it.
|
|
293
|
+
|
|
294
|
+
When a dedicated inference key is configured separately, use
|
|
295
|
+
`client.inference.listModels()` and `client.inference.retrieveModel(id)`: they
|
|
296
|
+
use the exact inference key and base URL selected for chat completions. Model
|
|
297
|
+
ids containing `author/name` are encoded safely. A partially readable list
|
|
298
|
+
includes `usf_unreachable_sources`; an unavailable source is not proof there
|
|
299
|
+
are no models in it. Neither SDK retries a failed inference request for you.
|
|
300
|
+
|
|
301
|
+
```typescript
|
|
302
|
+
const available = await client.inference.listModels();
|
|
303
|
+
for (const model of available.data) console.log(model.id);
|
|
304
|
+
if (available.usf_unreachable_sources?.length) {
|
|
305
|
+
console.log('Some model sources were unreachable; retry discovery before concluding they are empty.');
|
|
306
|
+
}
|
|
307
|
+
if (available.data.length) {
|
|
308
|
+
const detail = await client.inference.retrieveModel(available.data[0].id);
|
|
309
|
+
console.log(detail.id);
|
|
310
|
+
}
|
|
311
|
+
```
|
|
312
|
+
|
|
254
313
|
### Datasets
|
|
255
314
|
|
|
256
315
|
```typescript
|
package/dist/index.d.ts
CHANGED
|
@@ -36,7 +36,7 @@ import { Loop } from './resources/loop.js';
|
|
|
36
36
|
* SDK version. Sent as part of the User-Agent header.
|
|
37
37
|
* Must match package.json "version" -- enforced by a contract test.
|
|
38
38
|
*/
|
|
39
|
-
export declare const VERSION = "0.2.14-dev.
|
|
39
|
+
export declare const VERSION = "0.2.14-dev.247";
|
|
40
40
|
export declare class RunBiOS {
|
|
41
41
|
/** Search models, fetch configs, check adapter compatibility. */
|
|
42
42
|
readonly models: Models;
|
|
@@ -74,8 +74,8 @@ export { Integrations, type HuggingFaceIntegration, type IntegrationBrowseParams
|
|
|
74
74
|
export { Training } from './resources/training.js';
|
|
75
75
|
export { Wallet } from './resources/wallet.js';
|
|
76
76
|
export { GPU, type GPURecommendation } from './resources/gpu.js';
|
|
77
|
-
export { Inference, validateChatRequest, parseSSE, buildInferenceRequest, contextSizingBasis, type ContextSizingBasis, CONTEXT_DEFAULT_CEILING, CONTEXT_EDITABLE_FLOOR, type ChatMessage, type FunctionTool, type ChatCompletionParams, type ChatCompletionResponse, type ChatCompletionChunk, type ServerlessUsageWindow, type ServerlessTimeseriesMetric, type ServerlessWorkspaceLimits, type ServerlessUsageEnvelope, type ServerlessUsageOverviewResponse, type ServerlessUsageRowsResponse, type ServerlessUsageTimeseriesResponse, type ServerlessUsageDailyResponse, type ServerlessSavingsTotals, type ServerlessSavingsPeriod, type ServerlessUsageSavingsResponse, } from './resources/inference.js';
|
|
78
|
-
export type { BiOSConfig, PaginatedResponse, ApiErrorBody, AvailableGpuAlternative, CapacityMinimumRequirement, InferenceBookingAccepted, Model, ModelSearchParams, ModelSearchResponse, ModelDetailResponse, ModelConfig, Dataset, DatasetListParams, DatasetListResponse, DatasetUploadParams, DatasetPreview, DatasetPreviewParams, DatasetImportHFParams, DatasetRegisterHFParams, DatasetHubSearchParams, DatasetHubPreviewParams, DatasetValidation, DatasetFormatVariant, DatasetFormatSpec, DatasetFormatSpecs, DatasetStorageUsage, TrainingMethod, RLHFAlgorithm, AdapterType, TrainingJobStatus, TrainingStatusFilter, TrainingCreateParams, TrainingListParams, TrainingListResponse, TrainingJob, TrainingMetrics, TrainingResourceMetrics, TrainingDeviceMetrics, MetricPoint, MetricGraphConfig, TrainingCheckpoint, TrainingLogs, TrainingLogEntry, TrainingStopResponse, TrainingResumeResponse, CanonicalTrainingRequest, TrainingPreflightDataset, TrainingPreflightWarning, TrainingPreflightResponse, TrainingCapabilityChoice, TrainingConfigFieldCapability, TrainingCapabilities, GPUChoice, WalletBalance, Transaction, TransactionListResponse, TransactionListParams, GPUInfo, GPUPricingResponse, GPUOptionsParams, TrainingRecommendParams, TrainingAdvisorVerdict, TrainingAdvisorRecommendation, TrainingAdvisorUserShape, TrainingAdvisorJustification, TrainingAdvisorBasis, GPUOption, GPUOptionSuggestion, GPUOptionsResponse, InferenceStatus, InferenceCreateParams, InferenceUpdateParams, InferenceDeployment, InferenceDeploymentSummary, InferenceCreateResponse, InferenceListResponse, InferenceUpdateResponse, InferenceLifecycleResponse, InferenceDeleteResponse, InferenceGPUOptionMarket, InferenceGPUOption, InferenceGPUAlternative, InferenceGPUOptionsParams, InferenceGPUOptionsResponse, InferenceCanonicalRequest, InferencePreflightWarning, InferencePreflightResponse, AdapterCompatibility, AdapterCompatibilityResponse, AdapterCompatibilityParams, SupportedArchitecture, ArchitectureScope, SupportedArchitecturesResponse, ApiKey, ApiKeyScope, ApiKeyIntrospection, Organization, OrgMember, OrgInvite, Workspace, Integration, IntegrationCreateParams, StorageUsage, StorageObject, PromptTokensDetails, ChatCompletionUsage, TrainingCadence, TrainingCombinator, ServingKind, TrainingRulePausedReason, LoopTrainingMethod, TrainType, GradersScope, TrainingTrigger, TrainingRecipe, TrainingRunState, TrainingVerdict, TrainingDecision, NotifyState, EvaluationStatus, EvaluationItemStatus, EvaluationWinner, EvaluationWarningCode, ConsentVia, TrainingToolCall, TrainingMessage, TrainingRuleServing, TrainingGPURung, TrainingRule, TrainingRuleConsent, TrainingRulePreflightRefusal, TrainingRuleKeyRef, TrainingRuleKeyCheck, TrainingRulePreflight, TrainingRuleBuildSpec, TrainingRuleTriggerInput, TrainingRuleTrainingInput, TrainingRuleDeployInput, TrainingRuleMoneyInput, TrainingRuleEvaluationInput, TrainingRulePromotionInput, TrainingRuleAcceptTerms, TrainingRulePreflightRequest, TrainingRuleCreateRequest, TrainingRuleUpdateRequest, TrainingRuleConsentRequest, TrainingRuleListParams, TrainingRunPromoteRequest, TrainingRunRejectRequest, TrainingRunRollbackRequest, TrainingRunCancelRequest, TrainingRunListParams, TrainingRun, TrainingRunSummary, TrainingRunEvent, TrainingRunLinks, TrainingRunActions, EvaluationRef, EvaluationDecoding, EvaluationJudgeDimension, EvaluationDimensionScore, EvaluationGraderScore, EvaluationWarning, EvaluationMargin, JudgeAgreementDimension, JudgeAgreement, Evaluation, EvaluationItemSide, EvaluationItem, EvaluationItemListParams, JudgeAgreementParams, AgentSettings, AgentSettingsRequest, BenchmarkStatus, BenchmarkSourceKind, BenchmarkRunStatus, BenchmarkScoring, BenchmarkDimensionScore, Benchmark, BenchmarkItem, BenchmarkRun, BenchmarkHistoryPoint, BenchmarkSource, BenchmarkCreateParams, BenchmarkListParams, BenchmarkItemListParams, BenchmarkHistoryParams, BenchmarkRetireParams, TrainingRuleBenchmarkRequest, InferenceAlias, InferenceAliasRequest, TrainingRuleListResponse, TrainingRuleResponse, TrainingRuleMutationResponse, TrainingRuleDeleteResponse, TrainingRunListResponse, TrainingRunResponse, TrainingRunActionResponse, EvaluationItemsResponse, AgentSettingsResponse, InferenceAliasListResponse, InferenceAliasResponse, InferenceAliasDeleteResponse, BenchmarkListResponse, BenchmarkItemsResponse, BenchmarkHistoryResponse, TrainingRuleRunRefusalCode, TrainingRuleMonthlyLimitRefusal, LoopDatasetDeleteRefusalCode, PipelineStatus, PipelineRecipeExploration, PipelineActiveRun, PipelineVersion, Pipeline, PipelineListResponse, PipelineResponse, } from './types.js';
|
|
77
|
+
export { Inference, validateChatRequest, parseSSE, buildInferenceRequest, contextSizingBasis, type ContextSizingBasis, CONTEXT_DEFAULT_CEILING, CONTEXT_EDITABLE_FLOOR, type ChatMessage, type FunctionTool, type ChatCompletionParams, type ChatCompletionResponse, type ChatCompletionChunk, type InferenceVerbParams, type CompletionParams, type EmbeddingParams, type RerankParams, type AnthropicMessageParams, type ServerlessUsageWindow, type ServerlessTimeseriesMetric, type ServerlessWorkspaceLimits, type ServerlessUsageEnvelope, type ServerlessUsageOverviewResponse, type ServerlessUsageRowsResponse, type ServerlessUsageTimeseriesResponse, type ServerlessUsageDailyResponse, type ServerlessSavingsTotals, type ServerlessSavingsPeriod, type ServerlessUsageSavingsResponse, } from './resources/inference.js';
|
|
78
|
+
export type { BiOSConfig, PaginatedResponse, ApiErrorBody, AvailableGpuAlternative, CapacityMinimumRequirement, InferenceBookingAccepted, Model, ModelSearchParams, ModelSearchResponse, ModelDetailResponse, InferenceModel, InferenceModelListResponse, ModelConfig, Dataset, DatasetListParams, DatasetListResponse, DatasetUploadParams, DatasetPreview, DatasetPreviewParams, DatasetImportHFParams, DatasetRegisterHFParams, DatasetHubSearchParams, DatasetHubPreviewParams, DatasetValidation, DatasetFormatVariant, DatasetFormatSpec, DatasetFormatSpecs, DatasetStorageUsage, TrainingMethod, RLHFAlgorithm, AdapterType, TrainingJobStatus, TrainingStatusFilter, TrainingCreateParams, TrainingListParams, TrainingListResponse, TrainingJob, TrainingMetrics, TrainingResourceMetrics, TrainingDeviceMetrics, MetricPoint, MetricGraphConfig, TrainingCheckpoint, TrainingLogs, TrainingLogEntry, TrainingStopResponse, TrainingResumeResponse, CanonicalTrainingRequest, TrainingPreflightDataset, TrainingPreflightWarning, TrainingPreflightResponse, TrainingCapabilityChoice, TrainingConfigFieldCapability, TrainingCapabilities, GPUChoice, WalletBalance, Transaction, TransactionListResponse, TransactionListParams, GPUInfo, GPUPricingResponse, GPUOptionsParams, TrainingRecommendParams, TrainingAdvisorVerdict, TrainingAdvisorRecommendation, TrainingAdvisorUserShape, TrainingAdvisorJustification, TrainingAdvisorBasis, GPUOption, GPUOptionSuggestion, GPUOptionsResponse, InferenceStatus, InferenceCreateParams, InferenceUpdateParams, InferenceDeployment, InferenceDeploymentSummary, InferenceCreateResponse, InferenceListResponse, InferenceUpdateResponse, InferenceLifecycleResponse, InferenceDeleteResponse, InferenceGPUOptionMarket, InferenceGPUOption, InferenceGPUAlternative, InferenceGPUOptionsParams, InferenceGPUOptionsResponse, InferenceCanonicalRequest, InferencePreflightWarning, InferencePreflightResponse, AdapterCompatibility, AdapterCompatibilityResponse, AdapterCompatibilityParams, SupportedArchitecture, ArchitectureScope, SupportedArchitecturesResponse, ApiKey, ApiKeyScope, ApiKeyIntrospection, Organization, OrgMember, OrgInvite, Workspace, Integration, IntegrationCreateParams, StorageUsage, StorageObject, PromptTokensDetails, ChatCompletionUsage, TrainingCadence, TrainingCombinator, ServingKind, TrainingRulePausedReason, LoopTrainingMethod, TrainType, GradersScope, TrainingTrigger, TrainingRecipe, TrainingRunState, TrainingVerdict, TrainingDecision, NotifyState, EvaluationStatus, EvaluationItemStatus, EvaluationWinner, EvaluationWarningCode, ConsentVia, TrainingToolCall, TrainingMessage, TrainingRuleServing, TrainingGPURung, TrainingRule, TrainingRuleConsent, TrainingRulePreflightRefusal, TrainingRuleKeyRef, TrainingRuleKeyCheck, TrainingRulePreflight, TrainingRuleBuildSpec, TrainingRuleTriggerInput, TrainingRuleTrainingInput, TrainingRuleDeployInput, TrainingRuleMoneyInput, TrainingRuleEvaluationInput, TrainingRulePromotionInput, TrainingRuleAcceptTerms, TrainingRulePreflightRequest, TrainingRuleCreateRequest, TrainingRuleUpdateRequest, TrainingRuleConsentRequest, TrainingRuleListParams, TrainingRunPromoteRequest, TrainingRunRejectRequest, TrainingRunRollbackRequest, TrainingRunCancelRequest, TrainingRunListParams, TrainingRun, TrainingRunSummary, TrainingRunEvent, TrainingRunLinks, TrainingRunActions, EvaluationRef, EvaluationDecoding, EvaluationJudgeDimension, EvaluationDimensionScore, EvaluationGraderScore, EvaluationWarning, EvaluationMargin, JudgeAgreementDimension, JudgeAgreement, Evaluation, EvaluationItemSide, EvaluationItem, EvaluationItemListParams, JudgeAgreementParams, AgentSettings, AgentSettingsRequest, BenchmarkStatus, BenchmarkSourceKind, BenchmarkRunStatus, BenchmarkScoring, BenchmarkDimensionScore, Benchmark, BenchmarkItem, BenchmarkRun, BenchmarkHistoryPoint, BenchmarkSource, BenchmarkCreateParams, BenchmarkListParams, BenchmarkItemListParams, BenchmarkHistoryParams, BenchmarkRetireParams, TrainingRuleBenchmarkRequest, InferenceAlias, InferenceAliasRequest, TrainingRuleListResponse, TrainingRuleResponse, TrainingRuleMutationResponse, TrainingRuleDeleteResponse, TrainingRunListResponse, TrainingRunResponse, TrainingRunActionResponse, EvaluationItemsResponse, AgentSettingsResponse, InferenceAliasListResponse, InferenceAliasResponse, InferenceAliasDeleteResponse, BenchmarkListResponse, BenchmarkItemsResponse, BenchmarkHistoryResponse, TrainingRuleRunRefusalCode, TrainingRuleMonthlyLimitRefusal, LoopDatasetDeleteRefusalCode, PipelineStatus, PipelineRecipeExploration, PipelineActiveRun, PipelineVersion, Pipeline, PipelineListResponse, PipelineResponse, } from './types.js';
|
|
79
79
|
/** The closed set of run states a training run never leaves. */
|
|
80
80
|
export { TERMINAL_RUN_STATES } from './types.js';
|
|
81
81
|
/** The most versions a pipeline may be set to make. */
|
package/dist/index.js
CHANGED
|
@@ -36,7 +36,7 @@ import { Loop } from './resources/loop.js';
|
|
|
36
36
|
* SDK version. Sent as part of the User-Agent header.
|
|
37
37
|
* Must match package.json "version" -- enforced by a contract test.
|
|
38
38
|
*/
|
|
39
|
-
export const VERSION = '0.2.14-dev.
|
|
39
|
+
export const VERSION = '0.2.14-dev.247';
|
|
40
40
|
export class RunBiOS {
|
|
41
41
|
/** Search models, fetch configs, check adapter compatibility. */
|
|
42
42
|
models;
|
|
@@ -1,5 +1,5 @@
|
|
|
1
1
|
import type { HttpClient } from '../client.js';
|
|
2
|
-
import type { InferenceDeployment, InferenceDeploymentSummary, InferenceBookingAccepted, InferenceCreateParams, InferenceCreateResponse, InferenceDeleteResponse, InferenceGPUOptionsParams, InferenceGPUOptionsResponse, InferenceLifecycleResponse, InferenceListParams, InferenceListResponse, InferenceNotificationListResponse, InferencePreflightResponse, InferenceUpdateResponse, InferenceUpdateParams, InferenceAlias, InferenceAliasRequest, InferenceAliasDeleteResponse } from '../types.js';
|
|
2
|
+
import type { InferenceDeployment, InferenceDeploymentSummary, InferenceModel, InferenceModelListResponse, InferenceBookingAccepted, InferenceCreateParams, InferenceCreateResponse, InferenceDeleteResponse, InferenceGPUOptionsParams, InferenceGPUOptionsResponse, InferenceLifecycleResponse, InferenceListParams, InferenceListResponse, InferenceNotificationListResponse, InferencePreflightResponse, InferenceUpdateResponse, InferenceUpdateParams, InferenceAlias, InferenceAliasRequest, InferenceAliasDeleteResponse } from '../types.js';
|
|
3
3
|
/**
|
|
4
4
|
* Serving context-length policy, owned and enforced by the server. Mirrored
|
|
5
5
|
* here for documentation only -- never to pre-empt a server verdict.
|
|
@@ -78,6 +78,28 @@ export interface ChatCompletionParams extends Record<string, unknown> {
|
|
|
78
78
|
}
|
|
79
79
|
export type ChatCompletionResponse = Record<string, unknown>;
|
|
80
80
|
export type ChatCompletionChunk = Record<string, unknown>;
|
|
81
|
+
export interface InferenceVerbParams extends Record<string, unknown> {
|
|
82
|
+
model?: string;
|
|
83
|
+
inferenceKey?: string;
|
|
84
|
+
idempotencyKey?: string;
|
|
85
|
+
requestId?: string;
|
|
86
|
+
signal?: AbortSignal;
|
|
87
|
+
}
|
|
88
|
+
export interface CompletionParams extends InferenceVerbParams {
|
|
89
|
+
prompt: string | string[] | number[] | number[][];
|
|
90
|
+
}
|
|
91
|
+
export interface EmbeddingParams extends InferenceVerbParams {
|
|
92
|
+
input: string | string[] | number[] | number[][];
|
|
93
|
+
}
|
|
94
|
+
export interface RerankParams extends InferenceVerbParams {
|
|
95
|
+
query: string;
|
|
96
|
+
documents: Array<string | Record<string, unknown>>;
|
|
97
|
+
}
|
|
98
|
+
export interface AnthropicMessageParams extends InferenceVerbParams {
|
|
99
|
+
messages: Array<Record<string, unknown>>;
|
|
100
|
+
max_tokens: number;
|
|
101
|
+
anthropicVersion?: string;
|
|
102
|
+
}
|
|
81
103
|
export type ServerlessUsageWindow = '1h' | '24h' | '7d' | '30d' | '90d';
|
|
82
104
|
export type ServerlessTimeseriesMetric = 'requests' | 'tokens' | 'spend' | 'ttft_p50' | 'ttft_p95' | 'tps';
|
|
83
105
|
export interface ServerlessWorkspaceLimits {
|
|
@@ -151,8 +173,8 @@ export declare function validateChatRequest(body: Record<string, unknown>): void
|
|
|
151
173
|
export declare function parseSSE(body: ReadableStream<Uint8Array>): AsyncGenerator<string>;
|
|
152
174
|
/**
|
|
153
175
|
* Inference surface. Combines control-plane management of model-serving
|
|
154
|
-
* deployments (`/api/inference*`) with
|
|
155
|
-
*
|
|
176
|
+
* deployments (`/api/inference*`) with key-scoped inference on the supported
|
|
177
|
+
* `/v1` task routes. Requests are dispatched once; an idempotency header
|
|
156
178
|
* is forwarded but server-side replay is not assumed.
|
|
157
179
|
*/
|
|
158
180
|
export declare class Inference {
|
|
@@ -330,6 +352,18 @@ export declare class Inference {
|
|
|
330
352
|
getGPUOptions(params: InferenceGPUOptionsParams): Promise<InferenceGPUOptionsResponse>;
|
|
331
353
|
private prepare;
|
|
332
354
|
private abortContext;
|
|
355
|
+
private prepareVerb;
|
|
356
|
+
private sendVerb;
|
|
357
|
+
private modelRead;
|
|
358
|
+
private streamVerb;
|
|
359
|
+
completions(params: CompletionParams): Promise<Record<string, unknown>>;
|
|
360
|
+
embeddings(params: EmbeddingParams): Promise<Record<string, unknown>>;
|
|
361
|
+
rerank(params: RerankParams): Promise<Record<string, unknown>>;
|
|
362
|
+
messages(params: AnthropicMessageParams): Promise<Record<string, unknown>>;
|
|
363
|
+
streamCompletions(params: CompletionParams): AsyncGenerator<Record<string, unknown>>;
|
|
364
|
+
streamMessages(params: AnthropicMessageParams): AsyncGenerator<Record<string, unknown>>;
|
|
365
|
+
listModels(inferenceKey?: string): Promise<InferenceModelListResponse>;
|
|
366
|
+
retrieveModel(modelId: string, inferenceKey?: string): Promise<InferenceModel>;
|
|
333
367
|
chatCompletions(params: ChatCompletionParams): Promise<ChatCompletionResponse>;
|
|
334
368
|
streamChatCompletions(params: ChatCompletionParams): AsyncGenerator<ChatCompletionChunk>;
|
|
335
369
|
}
|
|
@@ -1,5 +1,5 @@
|
|
|
1
1
|
import { ApiError, GpuRejectionError, gpuRejectionCodeForReason, envApiKey, envBaseUrl, envInferenceKey } from '../client.js';
|
|
2
|
-
import { readNativeMaxContext } from './models.js';
|
|
2
|
+
import { asInferenceModel, asInferenceModelList, readNativeMaxContext } from './models.js';
|
|
3
3
|
import { normalizeGPUPlacement } from './gpu-priorities.js';
|
|
4
4
|
const FUNCTION_NAME = /^[A-Za-z0-9_-]{1,64}$/;
|
|
5
5
|
const ROLES = new Set(['system', 'developer', 'user', 'assistant', 'tool', 'function']);
|
|
@@ -318,8 +318,8 @@ async function apiError(response) {
|
|
|
318
318
|
}
|
|
319
319
|
/**
|
|
320
320
|
* Inference surface. Combines control-plane management of model-serving
|
|
321
|
-
* deployments (`/api/inference*`) with
|
|
322
|
-
*
|
|
321
|
+
* deployments (`/api/inference*`) with key-scoped inference on the supported
|
|
322
|
+
* `/v1` task routes. Requests are dispatched once; an idempotency header
|
|
323
323
|
* is forwarded but server-side replay is not assumed.
|
|
324
324
|
*/
|
|
325
325
|
export class Inference {
|
|
@@ -830,6 +830,170 @@ export class Inference {
|
|
|
830
830
|
remove: () => signal?.removeEventListener('abort', relay),
|
|
831
831
|
};
|
|
832
832
|
}
|
|
833
|
+
prepareVerb(params, stream, anthropic) {
|
|
834
|
+
const { inferenceKey, idempotencyKey, requestId, signal, anthropicVersion, stream: requestedStream, ...payload } = params;
|
|
835
|
+
if (requestedStream !== undefined)
|
|
836
|
+
throw new Error('Choose the streaming method instead of setting stream in a request body');
|
|
837
|
+
const key = inferenceKey || this.key;
|
|
838
|
+
if (!key)
|
|
839
|
+
throw new Error('an inferenceKey is required');
|
|
840
|
+
const headers = {
|
|
841
|
+
Authorization: `Bearer ${key}`,
|
|
842
|
+
Accept: stream ? 'text/event-stream' : 'application/json',
|
|
843
|
+
'Content-Type': 'application/json',
|
|
844
|
+
'X-Request-ID': requestId || crypto.randomUUID(),
|
|
845
|
+
};
|
|
846
|
+
if (idempotencyKey)
|
|
847
|
+
headers['Idempotency-Key'] = idempotencyKey;
|
|
848
|
+
if (anthropic) {
|
|
849
|
+
if (anthropicVersion !== undefined && typeof anthropicVersion !== 'string') {
|
|
850
|
+
throw new Error('anthropicVersion must be a string');
|
|
851
|
+
}
|
|
852
|
+
headers['Anthropic-Version'] = anthropicVersion || '2023-06-01';
|
|
853
|
+
}
|
|
854
|
+
return { body: stream ? { ...payload, stream: true } : payload, headers, signal };
|
|
855
|
+
}
|
|
856
|
+
async sendVerb(path, params, anthropic = false) {
|
|
857
|
+
const prepared = this.prepareVerb(params, false, anthropic);
|
|
858
|
+
const abort = this.abortContext(prepared.signal);
|
|
859
|
+
try {
|
|
860
|
+
const response = await fetch(`${this.baseUrl}${path}`, {
|
|
861
|
+
method: 'POST', headers: prepared.headers, body: JSON.stringify(prepared.body), signal: abort.controller.signal,
|
|
862
|
+
});
|
|
863
|
+
if (!response.ok)
|
|
864
|
+
throw await apiError(response);
|
|
865
|
+
let payload;
|
|
866
|
+
try {
|
|
867
|
+
payload = await response.json();
|
|
868
|
+
}
|
|
869
|
+
catch {
|
|
870
|
+
throw new ApiError(response.status, { error: 'Inference endpoint returned invalid JSON' });
|
|
871
|
+
}
|
|
872
|
+
if (!payload || typeof payload !== 'object' || Array.isArray(payload)) {
|
|
873
|
+
throw new ApiError(response.status, { error: 'Inference endpoint returned no response object' });
|
|
874
|
+
}
|
|
875
|
+
return payload;
|
|
876
|
+
}
|
|
877
|
+
catch (error) {
|
|
878
|
+
if (abort.controller.signal.aborted && !(error instanceof ApiError)) {
|
|
879
|
+
throw new ApiError(0, { error: String(abort.controller.signal.reason || 'Inference request aborted') });
|
|
880
|
+
}
|
|
881
|
+
throw error;
|
|
882
|
+
}
|
|
883
|
+
finally {
|
|
884
|
+
clearTimeout(abort.timeoutId);
|
|
885
|
+
abort.remove();
|
|
886
|
+
}
|
|
887
|
+
}
|
|
888
|
+
async modelRead(path, inferenceKey) {
|
|
889
|
+
const key = inferenceKey || this.key;
|
|
890
|
+
if (!key)
|
|
891
|
+
throw new Error('an inferenceKey is required');
|
|
892
|
+
const abort = this.abortContext();
|
|
893
|
+
try {
|
|
894
|
+
const response = await fetch(`${this.baseUrl}${path}`, {
|
|
895
|
+
method: 'GET',
|
|
896
|
+
headers: { Authorization: `Bearer ${key}`, Accept: 'application/json', 'X-Request-ID': crypto.randomUUID() },
|
|
897
|
+
signal: abort.controller.signal,
|
|
898
|
+
});
|
|
899
|
+
if (!response.ok)
|
|
900
|
+
throw await apiError(response);
|
|
901
|
+
try {
|
|
902
|
+
return await response.json();
|
|
903
|
+
}
|
|
904
|
+
catch {
|
|
905
|
+
throw new ApiError(response.status, { error: { code: 'INVALID_MODEL_RESPONSE', message: 'Inference model discovery returned invalid JSON.' } });
|
|
906
|
+
}
|
|
907
|
+
}
|
|
908
|
+
catch (error) {
|
|
909
|
+
if (abort.controller.signal.aborted && !(error instanceof ApiError)) {
|
|
910
|
+
throw new ApiError(0, { error: String(abort.controller.signal.reason || 'Inference model discovery aborted') });
|
|
911
|
+
}
|
|
912
|
+
throw error;
|
|
913
|
+
}
|
|
914
|
+
finally {
|
|
915
|
+
clearTimeout(abort.timeoutId);
|
|
916
|
+
abort.remove();
|
|
917
|
+
}
|
|
918
|
+
}
|
|
919
|
+
async *streamVerb(path, params, anthropic = false) {
|
|
920
|
+
const prepared = this.prepareVerb(params, true, anthropic);
|
|
921
|
+
const abort = this.abortContext(prepared.signal);
|
|
922
|
+
try {
|
|
923
|
+
const response = await fetch(`${this.baseUrl}${path}`, {
|
|
924
|
+
method: 'POST', headers: prepared.headers, body: JSON.stringify(prepared.body), signal: abort.controller.signal,
|
|
925
|
+
});
|
|
926
|
+
if (!response.ok)
|
|
927
|
+
throw await apiError(response);
|
|
928
|
+
const contentType = response.headers.get('content-type')?.toLowerCase() ?? '';
|
|
929
|
+
if (!contentType.includes('text/event-stream')) {
|
|
930
|
+
throw new ApiError(response.status, { error: `Expected text/event-stream, received ${contentType || 'no content type'}` });
|
|
931
|
+
}
|
|
932
|
+
if (!response.body)
|
|
933
|
+
throw new ApiError(0, { error: 'Inference stream returned no body' });
|
|
934
|
+
for await (const data of parseSSE(response.body)) {
|
|
935
|
+
if (data === '[DONE]') {
|
|
936
|
+
if (anthropic)
|
|
937
|
+
throw new ApiError(0, { error: 'Anthropic stream ended without message_stop' });
|
|
938
|
+
return;
|
|
939
|
+
}
|
|
940
|
+
let event;
|
|
941
|
+
try {
|
|
942
|
+
event = JSON.parse(data);
|
|
943
|
+
}
|
|
944
|
+
catch {
|
|
945
|
+
throw new ApiError(0, { error: 'Invalid JSON SSE event' });
|
|
946
|
+
}
|
|
947
|
+
if (!event || typeof event !== 'object' || Array.isArray(event)) {
|
|
948
|
+
throw new ApiError(0, { error: 'Inference stream returned an invalid event' });
|
|
949
|
+
}
|
|
950
|
+
if ('error' in event)
|
|
951
|
+
throw new ApiError(0, event);
|
|
952
|
+
yield event;
|
|
953
|
+
if (anthropic && event.type === 'message_stop')
|
|
954
|
+
return;
|
|
955
|
+
}
|
|
956
|
+
throw new ApiError(0, { error: 'Inference stream ended before completion marker' });
|
|
957
|
+
}
|
|
958
|
+
catch (error) {
|
|
959
|
+
if (abort.controller.signal.aborted && !(error instanceof ApiError)) {
|
|
960
|
+
throw new ApiError(0, { error: String(abort.controller.signal.reason || 'Inference request aborted') });
|
|
961
|
+
}
|
|
962
|
+
throw error;
|
|
963
|
+
}
|
|
964
|
+
finally {
|
|
965
|
+
abort.controller.abort('stream closed');
|
|
966
|
+
clearTimeout(abort.timeoutId);
|
|
967
|
+
abort.remove();
|
|
968
|
+
}
|
|
969
|
+
}
|
|
970
|
+
completions(params) {
|
|
971
|
+
return this.sendVerb('/v1/completions', params);
|
|
972
|
+
}
|
|
973
|
+
embeddings(params) {
|
|
974
|
+
return this.sendVerb('/v1/embeddings', params);
|
|
975
|
+
}
|
|
976
|
+
rerank(params) {
|
|
977
|
+
return this.sendVerb('/v1/rerank', params);
|
|
978
|
+
}
|
|
979
|
+
messages(params) {
|
|
980
|
+
return this.sendVerb('/v1/messages', params, true);
|
|
981
|
+
}
|
|
982
|
+
async *streamCompletions(params) {
|
|
983
|
+
yield* this.streamVerb('/v1/completions', params);
|
|
984
|
+
}
|
|
985
|
+
async *streamMessages(params) {
|
|
986
|
+
yield* this.streamVerb('/v1/messages', params, true);
|
|
987
|
+
}
|
|
988
|
+
async listModels(inferenceKey) {
|
|
989
|
+
return asInferenceModelList(await this.modelRead('/v1/models', inferenceKey));
|
|
990
|
+
}
|
|
991
|
+
async retrieveModel(modelId, inferenceKey) {
|
|
992
|
+
const id = typeof modelId === 'string' ? modelId.trim() : '';
|
|
993
|
+
if (!id)
|
|
994
|
+
throw new Error('RunBiOS: modelId is required to retrieve an inference model');
|
|
995
|
+
return asInferenceModel(await this.modelRead(`/v1/models/${encodeURIComponent(id)}`, inferenceKey));
|
|
996
|
+
}
|
|
833
997
|
async chatCompletions(params) {
|
|
834
998
|
const prepared = this.prepare(params, false);
|
|
835
999
|
const abort = this.abortContext(prepared.signal);
|
|
@@ -882,6 +1046,7 @@ export class Inference {
|
|
|
882
1046
|
}
|
|
883
1047
|
yield event;
|
|
884
1048
|
}
|
|
1049
|
+
throw new ApiError(0, { error: 'Inference stream ended before completion marker [DONE]' });
|
|
885
1050
|
}
|
|
886
1051
|
catch (error) {
|
|
887
1052
|
if (abort.controller.signal.aborted && !(error instanceof ApiError)) {
|
|
@@ -1,5 +1,5 @@
|
|
|
1
1
|
import type { HttpClient } from '../client.js';
|
|
2
|
-
import type { ModelDetailResponse, ModelSearchParams, ModelSearchResponse, ModelConfig, AdapterCompatibilityParams, AdapterCompatibilityResponse, ArchitectureScope, SupportedArchitecturesResponse } from '../types.js';
|
|
2
|
+
import type { InferenceModel, InferenceModelListResponse, ModelDetailResponse, ModelSearchParams, ModelSearchResponse, ModelConfig, AdapterCompatibilityParams, AdapterCompatibilityResponse, ArchitectureScope, SupportedArchitecturesResponse } from '../types.js';
|
|
3
3
|
/**
|
|
4
4
|
* Access the Run BiOS model catalog -- search models, fetch
|
|
5
5
|
* training-relevant configuration, and check adapter compatibility.
|
|
@@ -8,6 +8,8 @@ export declare class Models {
|
|
|
8
8
|
private readonly _http;
|
|
9
9
|
/** @internal */
|
|
10
10
|
constructor(_http: HttpClient);
|
|
11
|
+
list(): Promise<InferenceModelListResponse>;
|
|
12
|
+
retrieve(modelId: string): Promise<InferenceModel>;
|
|
11
13
|
/**
|
|
12
14
|
* Search the Run BiOS model catalog -- the platform's own hosted, verified
|
|
13
15
|
* models. Every result is mirrored in Run BiOS storage and can be trained and
|
|
@@ -98,6 +100,8 @@ export declare class Models {
|
|
|
98
100
|
scope?: ArchitectureScope;
|
|
99
101
|
}): Promise<SupportedArchitecturesResponse>;
|
|
100
102
|
}
|
|
103
|
+
export declare function asInferenceModelList(value: unknown): InferenceModelListResponse;
|
|
104
|
+
export declare function asInferenceModel(value: unknown): InferenceModel;
|
|
101
105
|
/** Registry detail path for an `author/name` catalog id, else undefined. @internal */
|
|
102
106
|
export declare function modelDetailPath(modelId: string): string | undefined;
|
|
103
107
|
/**
|
package/dist/resources/models.js
CHANGED
|
@@ -1,3 +1,4 @@
|
|
|
1
|
+
import { ApiError } from '../client.js';
|
|
1
2
|
/**
|
|
2
3
|
* Access the Run BiOS model catalog -- search models, fetch
|
|
3
4
|
* training-relevant configuration, and check adapter compatibility.
|
|
@@ -8,6 +9,15 @@ export class Models {
|
|
|
8
9
|
constructor(_http) {
|
|
9
10
|
this._http = _http;
|
|
10
11
|
}
|
|
12
|
+
async list() {
|
|
13
|
+
return asInferenceModelList(await this._http.fetchGet('/v1/models'));
|
|
14
|
+
}
|
|
15
|
+
async retrieve(modelId) {
|
|
16
|
+
const id = typeof modelId === 'string' ? modelId.trim() : '';
|
|
17
|
+
if (!id)
|
|
18
|
+
throw new Error('RunBiOS: modelId is required to retrieve an inference model');
|
|
19
|
+
return asInferenceModel(await this._http.fetchGet(`/v1/models/${encodeURIComponent(id)}`));
|
|
20
|
+
}
|
|
11
21
|
/**
|
|
12
22
|
* Search the Run BiOS model catalog -- the platform's own hosted, verified
|
|
13
23
|
* models. Every result is mirrored in Run BiOS storage and can be trained and
|
|
@@ -135,6 +145,22 @@ export class Models {
|
|
|
135
145
|
return this._http.fetchGet(`/api/public/serving-architectures?${q}`);
|
|
136
146
|
}
|
|
137
147
|
}
|
|
148
|
+
export function asInferenceModelList(value) {
|
|
149
|
+
const response = value && typeof value === 'object' ? value : null;
|
|
150
|
+
const data = response?.data;
|
|
151
|
+
if (response?.object !== 'list' || !Array.isArray(data) ||
|
|
152
|
+
!data.every(row => row && typeof row.id === 'string' && row.id.trim())) {
|
|
153
|
+
throw new ApiError(0, { error: { code: 'INVALID_MODELS_RESPONSE', message: 'Inference model discovery returned an invalid list; retry instead of treating it as empty.' } });
|
|
154
|
+
}
|
|
155
|
+
return value;
|
|
156
|
+
}
|
|
157
|
+
export function asInferenceModel(value) {
|
|
158
|
+
const model = value && typeof value === 'object' ? value : null;
|
|
159
|
+
if (model?.object !== 'model' || typeof model?.id !== 'string' || !model.id.trim()) {
|
|
160
|
+
throw new ApiError(0, { error: { code: 'INVALID_MODEL_RESPONSE', message: 'Inference model retrieval returned an invalid model; retry or list available models.' } });
|
|
161
|
+
}
|
|
162
|
+
return value;
|
|
163
|
+
}
|
|
138
164
|
/** Registry detail path for an `author/name` catalog id, else undefined. @internal */
|
|
139
165
|
export function modelDetailPath(modelId) {
|
|
140
166
|
const repo = (modelId || '').trim().replace(/^\/+|\/+$/g, '');
|
package/dist/types.d.ts
CHANGED
|
@@ -215,6 +215,22 @@ export interface ModelDetailResponse {
|
|
|
215
215
|
*/
|
|
216
216
|
primary_revision?: string;
|
|
217
217
|
}
|
|
218
|
+
export interface InferenceModel {
|
|
219
|
+
id: string;
|
|
220
|
+
object: string;
|
|
221
|
+
owned_by?: string;
|
|
222
|
+
created?: number;
|
|
223
|
+
[field: string]: unknown;
|
|
224
|
+
}
|
|
225
|
+
export interface InferenceModelListResponse {
|
|
226
|
+
object: string;
|
|
227
|
+
data: InferenceModel[];
|
|
228
|
+
usf_unreachable_sources?: Array<{
|
|
229
|
+
source: string;
|
|
230
|
+
reason: string;
|
|
231
|
+
}>;
|
|
232
|
+
[field: string]: unknown;
|
|
233
|
+
}
|
|
218
234
|
export interface ModelSearchParams {
|
|
219
235
|
/**
|
|
220
236
|
* Search text. Becomes the registry's `q` filter -- the ONLY search
|