runbios-sdk 0.2.13 → 0.2.14-dev.244
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +30 -1
- package/dist/client.js +1 -1
- package/dist/index.d.ts +5 -3
- package/dist/index.js +3 -1
- package/dist/resources/inference.d.ts +81 -0
- package/dist/resources/inference.js +61 -1
- package/dist/resources/loop.d.ts +53 -9
- package/dist/resources/loop.js +65 -8
- package/dist/types.d.ts +474 -7
- package/dist/types.js +3 -0
- package/package.json +1 -1
package/README.md
CHANGED
|
@@ -186,6 +186,35 @@ if (answer) {
|
|
|
186
186
|
Streaming billing is charged server-side on completed usage; the SDK only needs
|
|
187
187
|
to request `usage` where the endpoint exposes it (no client change).
|
|
188
188
|
|
|
189
|
+
#### Workspace serverless usage and limits (read-only)
|
|
190
|
+
|
|
191
|
+
A workspace-bound platform key or hosted OAuth grant with `analytics:read` can
|
|
192
|
+
read workspace RPM/spend settings and usage through the control-plane credential.
|
|
193
|
+
`serverless` alone allows inference but does not grant analytics. Set
|
|
194
|
+
`RUNBIOS_API_KEY` to an analytics:read workspace key (or pass it in the SDK
|
|
195
|
+
config). A dedicated serving key is not a control-plane credential. No
|
|
196
|
+
workspace id is accepted by these methods: the gateway resolves the bound workspace and checks access on
|
|
197
|
+
every call. Key creation/listing, per-key usage, org-wide totals and all RPM or
|
|
198
|
+
spend-cap writes remain in the JWT-authenticated console.
|
|
199
|
+
|
|
200
|
+
```typescript
|
|
201
|
+
const reporting = new RunBiOS({});
|
|
202
|
+
const limits = await reporting.inference.serverlessLimits();
|
|
203
|
+
const overview = await reporting.inference.serverlessUsageOverview('24h');
|
|
204
|
+
const byModel = await reporting.inference.serverlessUsageByModel('7d');
|
|
205
|
+
const requests = await reporting.inference.serverlessUsageRequests({ window: '7d', outcome: 'failed', limit: 25 });
|
|
206
|
+
const buckets = await reporting.inference.serverlessUsageTimeseries('spend', '7d');
|
|
207
|
+
const daily = await reporting.inference.serverlessUsageDaily(30);
|
|
208
|
+
const savings = await reporting.inference.serverlessUsageSavings('30d');
|
|
209
|
+
```
|
|
210
|
+
|
|
211
|
+
Windows are `1h`, `24h`, `7d`, `30d` or `90d`. Overview, model breakdown,
|
|
212
|
+
request logs and timeseries come from retained request details (about seven
|
|
213
|
+
days); `daily` uses full-history workspace counters over 1–92 UTC days and
|
|
214
|
+
includes failed admitted requests. The saved workspace RPM cap is a subdivision
|
|
215
|
+
of the org's tier allowance, not a claim about effective remaining RPM.
|
|
216
|
+
An incomplete 200 response is an error, never "zero usage" or "unlimited".
|
|
217
|
+
|
|
189
218
|
### Models
|
|
190
219
|
|
|
191
220
|
The catalog lists only models hosted on Run BiOS (the platform's own verified
|
|
@@ -491,7 +520,7 @@ try {
|
|
|
491
520
|
| `apiKey` | `RUNBIOS_API_KEY` env var (legacy `BIOS_API_KEY`) | Dashboard-issued API key (default `bios-`; custom and provider-shaped prefixes also work; legacy `usf-` keys remain valid) |
|
|
492
521
|
| `accessToken` | — | JWT access token |
|
|
493
522
|
| `orgId` | — | Organization ID (auto-resolved with API keys) |
|
|
494
|
-
| `workspaceId` | — | Workspace ID (auto-resolved with API keys) |
|
|
523
|
+
| `workspaceId` | — | Workspace ID (auto-resolved with workspace-bound API keys). An explicit ID must match the key's bound workspace; a workspace-less key can select a workspace it belongs to within its bound organization. |
|
|
495
524
|
| `baseUrl` | `RUNBIOS_BASE_URL` env var (legacy `BIOS_BASE_URL`), then `https://api.runbios.ai` | Canonical production hostname (release-gated; this documentation does not assert current availability). During prelaunch/dev, pass `https://api-dev.runbios.ai` explicitly. |
|
|
496
525
|
| `timeout` | `30000` | Request timeout in ms |
|
|
497
526
|
| `inferenceKey` | `RUNBIOS_INFERENCE_KEY` env var (legacy `BIOS_INFERENCE_KEY`), then `apiKey` | Key used by `client.inference` for `/v1` calls. A per-deployment `sk-bios-...` key, or the platform `apiKey` itself when it carries the serverless scope — you never pass the same key twice |
|
package/dist/client.js
CHANGED
|
@@ -340,7 +340,7 @@ export class HttpClient {
|
|
|
340
340
|
constructor(config) {
|
|
341
341
|
// Default host stays api.runbios.ai for now; cutover to api.runbios.ai is
|
|
342
342
|
// planned once its DNS exists.
|
|
343
|
-
this.baseUrl = (config.baseUrl || envBaseUrl() || 'https://api.runbios.ai').replace(/\/+$/, '');
|
|
343
|
+
this.baseUrl = (config.baseUrl || envBaseUrl() || 'https://api-dev.runbios.ai').replace(/\/+$/, '');
|
|
344
344
|
this.apiKey = config.apiKey ?? envApiKey();
|
|
345
345
|
this.accessToken = config.accessToken;
|
|
346
346
|
this.orgId = config.orgId;
|
package/dist/index.d.ts
CHANGED
|
@@ -36,7 +36,7 @@ import { Loop } from './resources/loop.js';
|
|
|
36
36
|
* SDK version. Sent as part of the User-Agent header.
|
|
37
37
|
* Must match package.json "version" -- enforced by a contract test.
|
|
38
38
|
*/
|
|
39
|
-
export declare const VERSION = "0.2.
|
|
39
|
+
export declare const VERSION = "0.2.14-dev.244";
|
|
40
40
|
export declare class RunBiOS {
|
|
41
41
|
/** Search models, fetch configs, check adapter compatibility. */
|
|
42
42
|
readonly models: Models;
|
|
@@ -74,7 +74,9 @@ export { Integrations, type HuggingFaceIntegration, type IntegrationBrowseParams
|
|
|
74
74
|
export { Training } from './resources/training.js';
|
|
75
75
|
export { Wallet } from './resources/wallet.js';
|
|
76
76
|
export { GPU, type GPURecommendation } from './resources/gpu.js';
|
|
77
|
-
export { Inference, validateChatRequest, parseSSE, buildInferenceRequest, contextSizingBasis, type ContextSizingBasis, CONTEXT_DEFAULT_CEILING, CONTEXT_EDITABLE_FLOOR, type ChatMessage, type FunctionTool, type ChatCompletionParams, type ChatCompletionResponse, type ChatCompletionChunk, } from './resources/inference.js';
|
|
78
|
-
export type { BiOSConfig, PaginatedResponse, ApiErrorBody, AvailableGpuAlternative, CapacityMinimumRequirement, InferenceBookingAccepted, Model, ModelSearchParams, ModelSearchResponse, ModelDetailResponse, ModelConfig, Dataset, DatasetListParams, DatasetListResponse, DatasetUploadParams, DatasetPreview, DatasetPreviewParams, DatasetImportHFParams, DatasetRegisterHFParams, DatasetHubSearchParams, DatasetHubPreviewParams, DatasetValidation, DatasetFormatVariant, DatasetFormatSpec, DatasetFormatSpecs, DatasetStorageUsage, TrainingMethod, RLHFAlgorithm, AdapterType, TrainingJobStatus, TrainingStatusFilter, TrainingCreateParams, TrainingListParams, TrainingListResponse, TrainingJob, TrainingMetrics, TrainingResourceMetrics, TrainingDeviceMetrics, MetricPoint, MetricGraphConfig, TrainingCheckpoint, TrainingLogs, TrainingLogEntry, TrainingStopResponse, TrainingResumeResponse, CanonicalTrainingRequest, TrainingPreflightDataset, TrainingPreflightWarning, TrainingPreflightResponse, TrainingCapabilityChoice, TrainingConfigFieldCapability, TrainingCapabilities, GPUChoice, WalletBalance, Transaction, TransactionListResponse, TransactionListParams, GPUInfo, GPUPricingResponse, GPUOptionsParams, TrainingRecommendParams, TrainingAdvisorVerdict, TrainingAdvisorRecommendation, TrainingAdvisorUserShape, TrainingAdvisorJustification, TrainingAdvisorBasis, GPUOption, GPUOptionSuggestion, GPUOptionsResponse, InferenceStatus, InferenceCreateParams, InferenceUpdateParams, InferenceDeployment, InferenceDeploymentSummary, InferenceCreateResponse, InferenceListResponse, InferenceUpdateResponse, InferenceLifecycleResponse, InferenceDeleteResponse, InferenceGPUOptionMarket, InferenceGPUOption, InferenceGPUAlternative, InferenceGPUOptionsParams, InferenceGPUOptionsResponse, InferenceCanonicalRequest, InferencePreflightWarning, InferencePreflightResponse, AdapterCompatibility, AdapterCompatibilityResponse, AdapterCompatibilityParams, SupportedArchitecture, ArchitectureScope, SupportedArchitecturesResponse, ApiKey, ApiKeyScope, ApiKeyIntrospection, Organization, OrgMember, OrgInvite, Workspace, Integration, IntegrationCreateParams, StorageUsage, StorageObject, PromptTokensDetails, ChatCompletionUsage, TrainingCadence, TrainingCombinator, ServingKind, TrainingRulePausedReason, LoopTrainingMethod, TrainType, GradersScope, TrainingTrigger, TrainingRunState, TrainingVerdict, TrainingDecision, NotifyState, EvaluationStatus, EvaluationItemStatus, EvaluationWinner, EvaluationWarningCode, ConsentVia, TrainingToolCall, TrainingMessage, TrainingRuleServing, TrainingGPURung, TrainingRule, TrainingRuleConsent, TrainingRulePreflightRefusal, TrainingRuleKeyRef, TrainingRuleKeyCheck, TrainingRulePreflight, TrainingRuleBuildSpec, TrainingRuleTriggerInput, TrainingRuleTrainingInput, TrainingRuleDeployInput, TrainingRuleMoneyInput, TrainingRuleEvaluationInput, TrainingRulePromotionInput, TrainingRuleAcceptTerms, TrainingRulePreflightRequest, TrainingRuleCreateRequest, TrainingRuleUpdateRequest, TrainingRuleConsentRequest, TrainingRuleListParams, TrainingRunPromoteRequest, TrainingRunRejectRequest, TrainingRunRollbackRequest, TrainingRunCancelRequest, TrainingRunListParams, TrainingRun, TrainingRunSummary, TrainingRunEvent, TrainingRunLinks, TrainingRunActions, EvaluationRef, EvaluationDecoding, EvaluationJudgeDimension, EvaluationDimensionScore, EvaluationGraderScore, EvaluationWarning, EvaluationMargin, JudgeAgreementDimension, JudgeAgreement, Evaluation, EvaluationItemSide, EvaluationItem, EvaluationItemListParams, JudgeAgreementParams, AgentSettings, AgentSettingsRequest, BenchmarkStatus, BenchmarkSourceKind, BenchmarkRunStatus, BenchmarkScoring, BenchmarkDimensionScore, Benchmark, BenchmarkItem, BenchmarkRun, BenchmarkHistoryPoint, BenchmarkSource, BenchmarkCreateParams, BenchmarkListParams, BenchmarkItemListParams, BenchmarkHistoryParams, BenchmarkRetireParams, TrainingRuleBenchmarkRequest, InferenceAlias, InferenceAliasRequest, TrainingRuleListResponse, TrainingRuleResponse, TrainingRuleMutationResponse, TrainingRuleDeleteResponse, TrainingRunListResponse, TrainingRunResponse, TrainingRunActionResponse, EvaluationItemsResponse, AgentSettingsResponse, InferenceAliasListResponse, InferenceAliasResponse, InferenceAliasDeleteResponse, BenchmarkListResponse, BenchmarkItemsResponse, BenchmarkHistoryResponse, } from './types.js';
|
|
77
|
+
export { Inference, validateChatRequest, parseSSE, buildInferenceRequest, contextSizingBasis, type ContextSizingBasis, CONTEXT_DEFAULT_CEILING, CONTEXT_EDITABLE_FLOOR, type ChatMessage, type FunctionTool, type ChatCompletionParams, type ChatCompletionResponse, type ChatCompletionChunk, type ServerlessUsageWindow, type ServerlessTimeseriesMetric, type ServerlessWorkspaceLimits, type ServerlessUsageEnvelope, type ServerlessUsageOverviewResponse, type ServerlessUsageRowsResponse, type ServerlessUsageTimeseriesResponse, type ServerlessUsageDailyResponse, type ServerlessSavingsTotals, type ServerlessSavingsPeriod, type ServerlessUsageSavingsResponse, } from './resources/inference.js';
|
|
78
|
+
export type { BiOSConfig, PaginatedResponse, ApiErrorBody, AvailableGpuAlternative, CapacityMinimumRequirement, InferenceBookingAccepted, Model, ModelSearchParams, ModelSearchResponse, ModelDetailResponse, ModelConfig, Dataset, DatasetListParams, DatasetListResponse, DatasetUploadParams, DatasetPreview, DatasetPreviewParams, DatasetImportHFParams, DatasetRegisterHFParams, DatasetHubSearchParams, DatasetHubPreviewParams, DatasetValidation, DatasetFormatVariant, DatasetFormatSpec, DatasetFormatSpecs, DatasetStorageUsage, TrainingMethod, RLHFAlgorithm, AdapterType, TrainingJobStatus, TrainingStatusFilter, TrainingCreateParams, TrainingListParams, TrainingListResponse, TrainingJob, TrainingMetrics, TrainingResourceMetrics, TrainingDeviceMetrics, MetricPoint, MetricGraphConfig, TrainingCheckpoint, TrainingLogs, TrainingLogEntry, TrainingStopResponse, TrainingResumeResponse, CanonicalTrainingRequest, TrainingPreflightDataset, TrainingPreflightWarning, TrainingPreflightResponse, TrainingCapabilityChoice, TrainingConfigFieldCapability, TrainingCapabilities, GPUChoice, WalletBalance, Transaction, TransactionListResponse, TransactionListParams, GPUInfo, GPUPricingResponse, GPUOptionsParams, TrainingRecommendParams, TrainingAdvisorVerdict, TrainingAdvisorRecommendation, TrainingAdvisorUserShape, TrainingAdvisorJustification, TrainingAdvisorBasis, GPUOption, GPUOptionSuggestion, GPUOptionsResponse, InferenceStatus, InferenceCreateParams, InferenceUpdateParams, InferenceDeployment, InferenceDeploymentSummary, InferenceCreateResponse, InferenceListResponse, InferenceUpdateResponse, InferenceLifecycleResponse, InferenceDeleteResponse, InferenceGPUOptionMarket, InferenceGPUOption, InferenceGPUAlternative, InferenceGPUOptionsParams, InferenceGPUOptionsResponse, InferenceCanonicalRequest, InferencePreflightWarning, InferencePreflightResponse, AdapterCompatibility, AdapterCompatibilityResponse, AdapterCompatibilityParams, SupportedArchitecture, ArchitectureScope, SupportedArchitecturesResponse, ApiKey, ApiKeyScope, ApiKeyIntrospection, Organization, OrgMember, OrgInvite, Workspace, Integration, IntegrationCreateParams, StorageUsage, StorageObject, PromptTokensDetails, ChatCompletionUsage, TrainingCadence, TrainingCombinator, ServingKind, TrainingRulePausedReason, LoopTrainingMethod, TrainType, GradersScope, TrainingTrigger, TrainingRecipe, TrainingRunState, TrainingVerdict, TrainingDecision, NotifyState, EvaluationStatus, EvaluationItemStatus, EvaluationWinner, EvaluationWarningCode, ConsentVia, TrainingToolCall, TrainingMessage, TrainingRuleServing, TrainingGPURung, TrainingRule, TrainingRuleConsent, TrainingRulePreflightRefusal, TrainingRuleKeyRef, TrainingRuleKeyCheck, TrainingRulePreflight, TrainingRuleBuildSpec, TrainingRuleTriggerInput, TrainingRuleTrainingInput, TrainingRuleDeployInput, TrainingRuleMoneyInput, TrainingRuleEvaluationInput, TrainingRulePromotionInput, TrainingRuleAcceptTerms, TrainingRulePreflightRequest, TrainingRuleCreateRequest, TrainingRuleUpdateRequest, TrainingRuleConsentRequest, TrainingRuleListParams, TrainingRunPromoteRequest, TrainingRunRejectRequest, TrainingRunRollbackRequest, TrainingRunCancelRequest, TrainingRunListParams, TrainingRun, TrainingRunSummary, TrainingRunEvent, TrainingRunLinks, TrainingRunActions, EvaluationRef, EvaluationDecoding, EvaluationJudgeDimension, EvaluationDimensionScore, EvaluationGraderScore, EvaluationWarning, EvaluationMargin, JudgeAgreementDimension, JudgeAgreement, Evaluation, EvaluationItemSide, EvaluationItem, EvaluationItemListParams, JudgeAgreementParams, AgentSettings, AgentSettingsRequest, BenchmarkStatus, BenchmarkSourceKind, BenchmarkRunStatus, BenchmarkScoring, BenchmarkDimensionScore, Benchmark, BenchmarkItem, BenchmarkRun, BenchmarkHistoryPoint, BenchmarkSource, BenchmarkCreateParams, BenchmarkListParams, BenchmarkItemListParams, BenchmarkHistoryParams, BenchmarkRetireParams, TrainingRuleBenchmarkRequest, InferenceAlias, InferenceAliasRequest, TrainingRuleListResponse, TrainingRuleResponse, TrainingRuleMutationResponse, TrainingRuleDeleteResponse, TrainingRunListResponse, TrainingRunResponse, TrainingRunActionResponse, EvaluationItemsResponse, AgentSettingsResponse, InferenceAliasListResponse, InferenceAliasResponse, InferenceAliasDeleteResponse, BenchmarkListResponse, BenchmarkItemsResponse, BenchmarkHistoryResponse, TrainingRuleRunRefusalCode, TrainingRuleMonthlyLimitRefusal, LoopDatasetDeleteRefusalCode, PipelineStatus, PipelineRecipeExploration, PipelineActiveRun, PipelineVersion, Pipeline, PipelineListResponse, PipelineResponse, } from './types.js';
|
|
79
79
|
/** The closed set of run states a training run never leaves. */
|
|
80
80
|
export { TERMINAL_RUN_STATES } from './types.js';
|
|
81
|
+
/** The most versions a pipeline may be set to make. */
|
|
82
|
+
export { PIPELINE_MAX_VERSIONS_CEILING } from './types.js';
|
package/dist/index.js
CHANGED
|
@@ -36,7 +36,7 @@ import { Loop } from './resources/loop.js';
|
|
|
36
36
|
* SDK version. Sent as part of the User-Agent header.
|
|
37
37
|
* Must match package.json "version" -- enforced by a contract test.
|
|
38
38
|
*/
|
|
39
|
-
export const VERSION = '0.2.
|
|
39
|
+
export const VERSION = '0.2.14-dev.244';
|
|
40
40
|
export class RunBiOS {
|
|
41
41
|
/** Search models, fetch configs, check adapter compatibility. */
|
|
42
42
|
models;
|
|
@@ -105,3 +105,5 @@ export { GPU } from './resources/gpu.js';
|
|
|
105
105
|
export { Inference, validateChatRequest, parseSSE, buildInferenceRequest, contextSizingBasis, CONTEXT_DEFAULT_CEILING, CONTEXT_EDITABLE_FLOOR, } from './resources/inference.js';
|
|
106
106
|
/** The closed set of run states a training run never leaves. */
|
|
107
107
|
export { TERMINAL_RUN_STATES } from './types.js';
|
|
108
|
+
/** The most versions a pipeline may be set to make. */
|
|
109
|
+
export { PIPELINE_MAX_VERSIONS_CEILING } from './types.js';
|
|
@@ -78,6 +78,75 @@ export interface ChatCompletionParams extends Record<string, unknown> {
|
|
|
78
78
|
}
|
|
79
79
|
export type ChatCompletionResponse = Record<string, unknown>;
|
|
80
80
|
export type ChatCompletionChunk = Record<string, unknown>;
|
|
81
|
+
export type ServerlessUsageWindow = '1h' | '24h' | '7d' | '30d' | '90d';
|
|
82
|
+
export type ServerlessTimeseriesMetric = 'requests' | 'tokens' | 'spend' | 'ttft_p50' | 'ttft_p95' | 'tps';
|
|
83
|
+
export interface ServerlessWorkspaceLimits {
|
|
84
|
+
workspace_id: string;
|
|
85
|
+
workspace_rpm_limit: number | null;
|
|
86
|
+
is_set: boolean;
|
|
87
|
+
monthly_spend_cap_cents: number | null;
|
|
88
|
+
spend_cap_is_set: boolean;
|
|
89
|
+
updated_at?: string;
|
|
90
|
+
}
|
|
91
|
+
export interface ServerlessUsageEnvelope {
|
|
92
|
+
window: string;
|
|
93
|
+
from: string;
|
|
94
|
+
to: string;
|
|
95
|
+
}
|
|
96
|
+
export interface ServerlessUsageOverviewResponse extends ServerlessUsageEnvelope {
|
|
97
|
+
overview: {
|
|
98
|
+
requests: number;
|
|
99
|
+
input_tokens: number;
|
|
100
|
+
output_tokens: number;
|
|
101
|
+
total_tokens: number;
|
|
102
|
+
spend_microcents: number;
|
|
103
|
+
[field: string]: unknown;
|
|
104
|
+
};
|
|
105
|
+
}
|
|
106
|
+
export interface ServerlessUsageRowsResponse extends ServerlessUsageEnvelope {
|
|
107
|
+
rows: Array<Record<string, unknown>>;
|
|
108
|
+
}
|
|
109
|
+
export interface ServerlessUsageTimeseriesResponse extends ServerlessUsageEnvelope {
|
|
110
|
+
metric: ServerlessTimeseriesMetric;
|
|
111
|
+
bucket_secs: number;
|
|
112
|
+
points: Array<{
|
|
113
|
+
bucket_ts: string;
|
|
114
|
+
value: number | null;
|
|
115
|
+
}>;
|
|
116
|
+
}
|
|
117
|
+
export interface ServerlessUsageDailyResponse {
|
|
118
|
+
days: number;
|
|
119
|
+
requests: number;
|
|
120
|
+
tokens_in: number;
|
|
121
|
+
tokens_out: number;
|
|
122
|
+
total_tokens: number;
|
|
123
|
+
spend_microcents: number;
|
|
124
|
+
points: Array<Record<string, unknown>>;
|
|
125
|
+
}
|
|
126
|
+
export interface ServerlessSavingsTotals {
|
|
127
|
+
savings_microcents: number;
|
|
128
|
+
standard_microcents: number;
|
|
129
|
+
charged_microcents: number;
|
|
130
|
+
requests: number;
|
|
131
|
+
}
|
|
132
|
+
export interface ServerlessSavingsPeriod {
|
|
133
|
+
you: {
|
|
134
|
+
savings_microcents: number;
|
|
135
|
+
since?: string | null;
|
|
136
|
+
};
|
|
137
|
+
workspace: {
|
|
138
|
+
savings_microcents: number;
|
|
139
|
+
since?: string | null;
|
|
140
|
+
};
|
|
141
|
+
}
|
|
142
|
+
export interface ServerlessUsageSavingsResponse extends ServerlessUsageEnvelope {
|
|
143
|
+
you: ServerlessSavingsTotals;
|
|
144
|
+
workspace: ServerlessSavingsTotals;
|
|
145
|
+
lifetime: ServerlessSavingsPeriod;
|
|
146
|
+
today: ServerlessSavingsPeriod;
|
|
147
|
+
by_model: Array<Record<string, unknown>>;
|
|
148
|
+
by_day: Array<Record<string, unknown>>;
|
|
149
|
+
}
|
|
81
150
|
export declare function validateChatRequest(body: Record<string, unknown>): void;
|
|
82
151
|
export declare function parseSSE(body: ReadableStream<Uint8Array>): AsyncGenerator<string>;
|
|
83
152
|
/**
|
|
@@ -98,6 +167,18 @@ export declare class Inference {
|
|
|
98
167
|
}, http?: HttpClient);
|
|
99
168
|
/** @internal Control-plane transport; present when constructed by the SDK client. */
|
|
100
169
|
private get http();
|
|
170
|
+
serverlessLimits(): Promise<ServerlessWorkspaceLimits>;
|
|
171
|
+
serverlessUsageOverview(window?: ServerlessUsageWindow): Promise<ServerlessUsageOverviewResponse>;
|
|
172
|
+
serverlessUsageByModel(window?: ServerlessUsageWindow): Promise<ServerlessUsageRowsResponse>;
|
|
173
|
+
serverlessUsageRequests(options?: {
|
|
174
|
+
window?: ServerlessUsageWindow;
|
|
175
|
+
outcome?: 'all' | 'ok' | 'failed' | 'rejected';
|
|
176
|
+
requestId?: string;
|
|
177
|
+
limit?: number;
|
|
178
|
+
}): Promise<ServerlessUsageRowsResponse>;
|
|
179
|
+
serverlessUsageTimeseries(metric: ServerlessTimeseriesMetric, window?: ServerlessUsageWindow): Promise<ServerlessUsageTimeseriesResponse>;
|
|
180
|
+
serverlessUsageDaily(days?: number): Promise<ServerlessUsageDailyResponse>;
|
|
181
|
+
serverlessUsageSavings(window?: ServerlessUsageWindow): Promise<ServerlessUsageSavingsResponse>;
|
|
101
182
|
/** Side-effect-free validation with authoritative stock, prices, alternatives, and hold terms. */
|
|
102
183
|
preflight(params: InferenceCreateParams): Promise<InferencePreflightResponse>;
|
|
103
184
|
/**
|
|
@@ -135,6 +135,16 @@ function buildInferenceRequest(params) {
|
|
|
135
135
|
body.max_price_hour_cents = params.maxPriceHourCents;
|
|
136
136
|
return body;
|
|
137
137
|
}
|
|
138
|
+
function serverlessRead(value, valid) {
|
|
139
|
+
const body = value && typeof value === 'object' && !Array.isArray(value) ? value : null;
|
|
140
|
+
if (!body || !valid(body)) {
|
|
141
|
+
throw new ApiError(0, { error: { code: 'INVALID_SERVERLESS_RESPONSE', message: 'Serverless data was incomplete; do not treat it as empty usage or an unset limit.' } });
|
|
142
|
+
}
|
|
143
|
+
return value;
|
|
144
|
+
}
|
|
145
|
+
function hasUsageWindow(body) {
|
|
146
|
+
return typeof body.window === 'string' && typeof body.from === 'string' && typeof body.to === 'string';
|
|
147
|
+
}
|
|
138
148
|
export function validateChatRequest(body) {
|
|
139
149
|
const messages = body.messages;
|
|
140
150
|
if (!Array.isArray(messages) || messages.length === 0) {
|
|
@@ -326,7 +336,7 @@ export class Inference {
|
|
|
326
336
|
this.key = config.inferenceKey || envInferenceKey() || envApiKey();
|
|
327
337
|
// Same default host as HttpClient (api.runbios.ai); api.runbios.ai cutover
|
|
328
338
|
// is planned once its DNS exists — update both call sites together.
|
|
329
|
-
this.baseUrl = (config.baseUrl || envBaseUrl() || 'https://api.runbios.ai').replace(/\/+$/, '');
|
|
339
|
+
this.baseUrl = (config.baseUrl || envBaseUrl() || 'https://api-dev.runbios.ai').replace(/\/+$/, '');
|
|
330
340
|
this.timeout = config.timeout ?? 900_000;
|
|
331
341
|
this._http = http;
|
|
332
342
|
}
|
|
@@ -337,6 +347,56 @@ export class Inference {
|
|
|
337
347
|
}
|
|
338
348
|
return this._http;
|
|
339
349
|
}
|
|
350
|
+
serverlessLimits() {
|
|
351
|
+
return this.http.fetchGet('/api/serverless/limits').then(value => serverlessRead(value, body => typeof body.workspace_id === 'string' && body.workspace_id.length > 0 &&
|
|
352
|
+
typeof body.is_set === 'boolean' && typeof body.spend_cap_is_set === 'boolean' &&
|
|
353
|
+
(body.workspace_rpm_limit === null || typeof body.workspace_rpm_limit === 'number') &&
|
|
354
|
+
(body.monthly_spend_cap_cents === null || typeof body.monthly_spend_cap_cents === 'number') &&
|
|
355
|
+
body.is_set === (body.workspace_rpm_limit !== null) &&
|
|
356
|
+
body.spend_cap_is_set === (body.monthly_spend_cap_cents !== null)));
|
|
357
|
+
}
|
|
358
|
+
serverlessUsageOverview(window = '24h') {
|
|
359
|
+
return this.http.fetchGet(`/api/serverless/usage/overview?window=${encodeURIComponent(window)}`).then(value => serverlessRead(value, body => hasUsageWindow(body) && body.overview !== null && typeof body.overview === 'object' &&
|
|
360
|
+
['requests', 'input_tokens', 'output_tokens', 'total_tokens', 'spend_microcents'].every(field => typeof body.overview[field] === 'number')));
|
|
361
|
+
}
|
|
362
|
+
serverlessUsageByModel(window = '24h') {
|
|
363
|
+
return this.http.fetchGet(`/api/serverless/usage/by-model?window=${encodeURIComponent(window)}`).then(value => serverlessRead(value, body => hasUsageWindow(body) && Array.isArray(body.rows)));
|
|
364
|
+
}
|
|
365
|
+
async serverlessUsageRequests(options = {}) {
|
|
366
|
+
const params = new URLSearchParams({ window: options.window ?? '24h' });
|
|
367
|
+
if (options.outcome)
|
|
368
|
+
params.set('outcome', options.outcome);
|
|
369
|
+
if (options.requestId)
|
|
370
|
+
params.set('request_id', options.requestId);
|
|
371
|
+
if (options.limit !== undefined) {
|
|
372
|
+
if (!Number.isInteger(options.limit) || options.limit < 1 || options.limit > 500)
|
|
373
|
+
throw new RangeError('limit must be an integer from 1 to 500');
|
|
374
|
+
params.set('limit', String(options.limit));
|
|
375
|
+
}
|
|
376
|
+
return this.http.fetchGet(`/api/serverless/usage/requests?${params}`).then(value => serverlessRead(value, body => hasUsageWindow(body) && Array.isArray(body.rows)));
|
|
377
|
+
}
|
|
378
|
+
serverlessUsageTimeseries(metric, window = '24h') {
|
|
379
|
+
const params = new URLSearchParams({ metric, window });
|
|
380
|
+
return this.http.fetchGet(`/api/serverless/usage/timeseries?${params}`).then(value => serverlessRead(value, body => hasUsageWindow(body) && body.metric === metric && typeof body.bucket_secs === 'number' && Array.isArray(body.points)));
|
|
381
|
+
}
|
|
382
|
+
serverlessUsageDaily(days = 30) {
|
|
383
|
+
return this.http.fetchGet(`/api/serverless/usage/daily?days=${encodeURIComponent(String(days))}`).then(value => serverlessRead(value, body => typeof body.days === 'number' && Number.isInteger(body.days) && body.days > 0 && Array.isArray(body.points) &&
|
|
384
|
+
['requests', 'tokens_in', 'tokens_out', 'total_tokens', 'spend_microcents'].every(field => typeof body[field] === 'number')));
|
|
385
|
+
}
|
|
386
|
+
serverlessUsageSavings(window = '30d') {
|
|
387
|
+
const record = (value) => value && typeof value === 'object' && !Array.isArray(value) ? value : null;
|
|
388
|
+
const totals = (value) => {
|
|
389
|
+
const row = record(value);
|
|
390
|
+
return !!row && ['savings_microcents', 'standard_microcents', 'charged_microcents', 'requests']
|
|
391
|
+
.every(field => typeof row[field] === 'number');
|
|
392
|
+
};
|
|
393
|
+
const period = (value) => {
|
|
394
|
+
const row = record(value);
|
|
395
|
+
return !!row && ['you', 'workspace'].every(scope => typeof record(row[scope])?.savings_microcents === 'number');
|
|
396
|
+
};
|
|
397
|
+
return this.http.fetchGet(`/api/serverless/usage/savings?window=${encodeURIComponent(window)}`).then(value => serverlessRead(value, body => hasUsageWindow(body) && totals(body.you) && totals(body.workspace) &&
|
|
398
|
+
period(body.lifetime) && period(body.today) && Array.isArray(body.by_model) && Array.isArray(body.by_day)));
|
|
399
|
+
}
|
|
340
400
|
// --------------------------------------------------------------------------
|
|
341
401
|
// Control-plane deployment management — `/api/inference*`
|
|
342
402
|
// --------------------------------------------------------------------------
|
package/dist/resources/loop.d.ts
CHANGED
|
@@ -1,5 +1,5 @@
|
|
|
1
1
|
import type { HttpClient } from '../client.js';
|
|
2
|
-
import type { LoopTrace, LoopTraceListResponse, LoopTraceListParams, LoopCaptureParams, LoopCaptureResult, LoopImportParams, LoopImportResult, LoopSignal, LoopSignalParams, LoopCandidate, LoopCandidateParams, LoopGrader, LoopGraderParams, LoopLabel, LoopLabelCount, LoopGradeResult, LoopDataset, LoopDatasetListParams, LoopDatasetItem, LoopDatasetCreateParams, LoopConfig, LoopConfigParams, LoopBuildRule, LoopBuildRuleParams, LoopBuildRuleUpdateParams, LoopStats, LoopJudge, LoopJudgeParams, LoopAgentCredential, LoopAgentStatus, LoopSampleRun, LoopSampleRunParams, LoopJudgeRun, LoopJudgeRunItems, LoopJudgeRunItemStatus, LoopJudgeWork, LoopJudgeVerdict, LoopJudgeVerdictResult, TrainingRule, TrainingRulePreflight, TrainingRulePreflightRequest, TrainingRuleCreateRequest, TrainingRuleUpdateRequest, TrainingRuleConsentRequest, TrainingRuleListParams, TrainingRuleListResponse, TrainingRuleResponse, TrainingRuleMutationResponse, TrainingRuleDeleteResponse, TrainingRun, TrainingRunListParams, TrainingRunListResponse, TrainingRunResponse, TrainingRunActionResponse, TrainingRunPromoteRequest, TrainingRunRejectRequest, TrainingRunRollbackRequest, TrainingRunCancelRequest, Evaluation, EvaluationItemListParams, EvaluationItemsResponse, JudgeAgreement, JudgeAgreementParams, Benchmark, BenchmarkCreateParams, BenchmarkHistoryParams, BenchmarkHistoryResponse, BenchmarkItemListParams, BenchmarkItemsResponse, BenchmarkListParams, BenchmarkListResponse, BenchmarkRetireParams, BenchmarkRun, TrainingRuleBenchmarkRequest, AgentSettings, AgentSettingsRequest } from '../types.js';
|
|
2
|
+
import type { LoopTrace, LoopTraceListResponse, LoopTraceListParams, LoopCaptureParams, LoopCaptureResult, LoopImportParams, LoopImportResult, LoopSignal, LoopSignalParams, LoopCandidate, LoopCandidateParams, LoopGrader, LoopGraderParams, LoopLabel, LoopLabelCount, LoopGradeResult, LoopDataset, LoopDatasetListParams, LoopDatasetItem, LoopDatasetCreateParams, LoopConfig, LoopConfigParams, LoopBuildRule, LoopBuildRuleParams, LoopBuildRuleUpdateParams, LoopStats, LoopJudge, LoopJudgeParams, LoopAgentCredential, LoopAgentStatus, LoopSampleRun, LoopSampleRunParams, LoopJudgeRun, LoopJudgeRunItems, LoopJudgeRunItemStatus, LoopJudgeWork, LoopJudgeVerdict, LoopJudgeVerdictResult, TrainingRule, TrainingRulePreflight, TrainingRulePreflightRequest, TrainingRuleCreateRequest, TrainingRuleUpdateRequest, TrainingRuleConsentRequest, TrainingRuleListParams, TrainingRuleListResponse, TrainingRuleResponse, TrainingRuleMutationResponse, TrainingRuleDeleteResponse, TrainingRun, TrainingRunListParams, TrainingRunListResponse, TrainingRunResponse, TrainingRunActionResponse, TrainingRunPromoteRequest, TrainingRunRejectRequest, TrainingRunRollbackRequest, TrainingRunCancelRequest, Evaluation, EvaluationItemListParams, EvaluationItemsResponse, JudgeAgreement, JudgeAgreementParams, Benchmark, BenchmarkCreateParams, BenchmarkHistoryParams, BenchmarkHistoryResponse, BenchmarkItemListParams, BenchmarkItemsResponse, BenchmarkListParams, BenchmarkListResponse, BenchmarkRetireParams, BenchmarkRun, TrainingRuleBenchmarkRequest, AgentSettings, AgentSettingsRequest, Pipeline, PipelineListResponse } from '../types.js';
|
|
3
3
|
/**
|
|
4
4
|
* The Conscious Loop -- capture what your model was asked and answered, record
|
|
5
5
|
* whether it was right, and turn those judgements into training data.
|
|
@@ -335,7 +335,10 @@ export declare class Loop {
|
|
|
335
335
|
* Delete a training set.
|
|
336
336
|
*
|
|
337
337
|
* The conversations it was built from are untouched -- a set is a selection,
|
|
338
|
-
* and discarding the selection must not discard the evidence.
|
|
338
|
+
* and discarding the selection must not discard the evidence. A set a
|
|
339
|
+
* training run was trained on is refused `409 DATASET_IN_USE` and kept, as
|
|
340
|
+
* the record of what that run learned from. See
|
|
341
|
+
* `LoopDatasetDeleteRefusalCode`.
|
|
339
342
|
*/
|
|
340
343
|
deleteDataset(id: string): Promise<{
|
|
341
344
|
deleted: boolean;
|
|
@@ -614,11 +617,12 @@ export declare class Loop {
|
|
|
614
617
|
*/
|
|
615
618
|
listTrainingRules(params?: TrainingRuleListParams): Promise<TrainingRuleListResponse>;
|
|
616
619
|
/**
|
|
617
|
-
* Read one rule with its recent runs, what
|
|
618
|
-
* far its judge agrees with your own reviewers.
|
|
620
|
+
* Read one rule with its recent runs, what its runs this month have cost at
|
|
621
|
+
* most, and how far its judge agrees with your own reviewers.
|
|
619
622
|
*
|
|
620
623
|
* Returned whole rather than unwrapped to the rule: `month_spent_cents` is
|
|
621
|
-
* the number that says whether the monthly ceiling is about to stop it.
|
|
624
|
+
* the number that says whether the monthly ceiling is about to stop it. It
|
|
625
|
+
* is an upper bound, not an exact spend: see `Pipeline.month_spent_cents`.
|
|
622
626
|
*/
|
|
623
627
|
getTrainingRule(id: string): Promise<TrainingRuleResponse>;
|
|
624
628
|
/**
|
|
@@ -626,8 +630,11 @@ export declare class Loop {
|
|
|
626
630
|
* that can be cleared.
|
|
627
631
|
*
|
|
628
632
|
* A money-bearing change -- a ceiling, a model, a GPU ladder, the promotion
|
|
629
|
-
* policy
|
|
630
|
-
*
|
|
633
|
+
* policy, or changing `explore_recipes` in EITHER direction -- bumps
|
|
634
|
+
* `revision`, clears the recorded consent and STOPS the rule firing until
|
|
635
|
+
* someone accepts the new amounts. Turning recipe variants OFF does this
|
|
636
|
+
* too: the pipeline then makes no version at all, variant or not, until the
|
|
637
|
+
* terms are accepted again. The reply says so in
|
|
631
638
|
* `consent_required`, and carries a fresh `preflight` with the new figures.
|
|
632
639
|
* Pass `expected_revision` to be refused with `REVISION_MISMATCH` rather
|
|
633
640
|
* than overwrite an edit somebody else made in the meantime.
|
|
@@ -654,8 +661,17 @@ export declare class Loop {
|
|
|
654
661
|
* Fire a rule now, without waiting for its cadence.
|
|
655
662
|
*
|
|
656
663
|
* Bypasses the schedule and `min_new_rows` only. The row floors, the money
|
|
657
|
-
* ceilings
|
|
658
|
-
*
|
|
664
|
+
* ceilings, the consent, the version limit and the monthly limit all still
|
|
665
|
+
* apply, so this can answer `409 CONSENT_REQUIRED`, `409 RULE_PAUSED`,
|
|
666
|
+
* `409 RUN_ACTIVE`, `409 VERSION_LIMIT_REACHED` (the pipeline has made
|
|
667
|
+
* `max_versions` versions), `409 MONTHLY_LIMIT_REACHED` (the most this
|
|
668
|
+
* month's runs can have cost plus the most one run may cost would pass the
|
|
669
|
+
* monthly limit; the error
|
|
670
|
+
* body carries `month_spent_cents`, `monthly_ceiling_cents`, `run_max_cents`
|
|
671
|
+
* and `resumes_at`), `409 AGENT_OFF` (the workspace's Conscious Loop agent
|
|
672
|
+
* is not turned on, so a trained model could not be compared) or
|
|
673
|
+
* `422 NOT_ENOUGH_ROWS` with the counts it needed. See
|
|
674
|
+
* `TrainingRuleRunRefusalCode`.
|
|
659
675
|
*/
|
|
660
676
|
runTrainingRule(id: string): Promise<TrainingRun>;
|
|
661
677
|
/**
|
|
@@ -710,6 +726,34 @@ export declare class Loop {
|
|
|
710
726
|
* cancelling a training job does not refund the hours it burned.
|
|
711
727
|
*/
|
|
712
728
|
cancelTrainingRun(id: string, params?: TrainingRunCancelRequest): Promise<TrainingRunActionResponse>;
|
|
729
|
+
/**
|
|
730
|
+
* Every training rule in the workspace, seen as the series of versions it
|
|
731
|
+
* produced: which exist, which one serves (`champion_version`), how each did
|
|
732
|
+
* against the champion of its day and on the standing benchmark, the run in
|
|
733
|
+
* flight and which version it will be, and what the pipeline is waiting for.
|
|
734
|
+
*
|
|
735
|
+
* Read-only. Every decision stays on the route that owns it -- promote,
|
|
736
|
+
* reject and roll back on the run, the version limit and `explore_recipes`
|
|
737
|
+
* on the rule.
|
|
738
|
+
*
|
|
739
|
+
* RETURNED WITH ITS ENVELOPE, because the list is not always all of them:
|
|
740
|
+
* the service returns the newest 100, `total` is every pipeline in the
|
|
741
|
+
* workspace and `truncated` is true when more exist than were returned. A
|
|
742
|
+
* caller that shows `pipelines` alone presents a short list as the whole of
|
|
743
|
+
* it. The rest are reached through `listTrainingRules`, which pages.
|
|
744
|
+
*/
|
|
745
|
+
listPipelines(): Promise<PipelineListResponse>;
|
|
746
|
+
/**
|
|
747
|
+
* One pipeline, by its training rule's id.
|
|
748
|
+
*
|
|
749
|
+
* `status` says whether it is the platform working (`running`), the member
|
|
750
|
+
* who has to act (`needs_review`, `needs_funds`), or nothing at all until
|
|
751
|
+
* somebody does (`paused`, with `paused_reason` or `needs_consent` saying
|
|
752
|
+
* why; `complete` at `max_versions`). `month_spent_cents` is the figure the
|
|
753
|
+
* monthly limit is enforced against: what runs started this month have cost
|
|
754
|
+
* at most, not an exact spend.
|
|
755
|
+
*/
|
|
756
|
+
getPipeline(id: string): Promise<Pipeline>;
|
|
713
757
|
/**
|
|
714
758
|
* The comparison behind a verdict: both models on the same held-out rows,
|
|
715
759
|
* with identical decoding, judge and grader names resolved.
|
package/dist/resources/loop.js
CHANGED
|
@@ -440,7 +440,10 @@ export class Loop {
|
|
|
440
440
|
* Delete a training set.
|
|
441
441
|
*
|
|
442
442
|
* The conversations it was built from are untouched -- a set is a selection,
|
|
443
|
-
* and discarding the selection must not discard the evidence.
|
|
443
|
+
* and discarding the selection must not discard the evidence. A set a
|
|
444
|
+
* training run was trained on is refused `409 DATASET_IN_USE` and kept, as
|
|
445
|
+
* the record of what that run learned from. See
|
|
446
|
+
* `LoopDatasetDeleteRefusalCode`.
|
|
444
447
|
*/
|
|
445
448
|
async deleteDataset(id) {
|
|
446
449
|
return this._http.fetchDelete(`/api/loop/datasets/${encodeURIComponent(id)}`);
|
|
@@ -846,11 +849,12 @@ export class Loop {
|
|
|
846
849
|
return this._http.fetchGet(`/api/loop/training-rules${qs ? `?${qs}` : ''}`);
|
|
847
850
|
}
|
|
848
851
|
/**
|
|
849
|
-
* Read one rule with its recent runs, what
|
|
850
|
-
* far its judge agrees with your own reviewers.
|
|
852
|
+
* Read one rule with its recent runs, what its runs this month have cost at
|
|
853
|
+
* most, and how far its judge agrees with your own reviewers.
|
|
851
854
|
*
|
|
852
855
|
* Returned whole rather than unwrapped to the rule: `month_spent_cents` is
|
|
853
|
-
* the number that says whether the monthly ceiling is about to stop it.
|
|
856
|
+
* the number that says whether the monthly ceiling is about to stop it. It
|
|
857
|
+
* is an upper bound, not an exact spend: see `Pipeline.month_spent_cents`.
|
|
854
858
|
*/
|
|
855
859
|
async getTrainingRule(id) {
|
|
856
860
|
return this._http.fetchGet(`/api/loop/training-rules/${encodeURIComponent(id)}`);
|
|
@@ -860,8 +864,11 @@ export class Loop {
|
|
|
860
864
|
* that can be cleared.
|
|
861
865
|
*
|
|
862
866
|
* A money-bearing change -- a ceiling, a model, a GPU ladder, the promotion
|
|
863
|
-
* policy
|
|
864
|
-
*
|
|
867
|
+
* policy, or changing `explore_recipes` in EITHER direction -- bumps
|
|
868
|
+
* `revision`, clears the recorded consent and STOPS the rule firing until
|
|
869
|
+
* someone accepts the new amounts. Turning recipe variants OFF does this
|
|
870
|
+
* too: the pipeline then makes no version at all, variant or not, until the
|
|
871
|
+
* terms are accepted again. The reply says so in
|
|
865
872
|
* `consent_required`, and carries a fresh `preflight` with the new figures.
|
|
866
873
|
* Pass `expected_revision` to be refused with `REVISION_MISMATCH` rather
|
|
867
874
|
* than overwrite an edit somebody else made in the meantime.
|
|
@@ -893,8 +900,17 @@ export class Loop {
|
|
|
893
900
|
* Fire a rule now, without waiting for its cadence.
|
|
894
901
|
*
|
|
895
902
|
* Bypasses the schedule and `min_new_rows` only. The row floors, the money
|
|
896
|
-
* ceilings
|
|
897
|
-
*
|
|
903
|
+
* ceilings, the consent, the version limit and the monthly limit all still
|
|
904
|
+
* apply, so this can answer `409 CONSENT_REQUIRED`, `409 RULE_PAUSED`,
|
|
905
|
+
* `409 RUN_ACTIVE`, `409 VERSION_LIMIT_REACHED` (the pipeline has made
|
|
906
|
+
* `max_versions` versions), `409 MONTHLY_LIMIT_REACHED` (the most this
|
|
907
|
+
* month's runs can have cost plus the most one run may cost would pass the
|
|
908
|
+
* monthly limit; the error
|
|
909
|
+
* body carries `month_spent_cents`, `monthly_ceiling_cents`, `run_max_cents`
|
|
910
|
+
* and `resumes_at`), `409 AGENT_OFF` (the workspace's Conscious Loop agent
|
|
911
|
+
* is not turned on, so a trained model could not be compared) or
|
|
912
|
+
* `422 NOT_ENOUGH_ROWS` with the counts it needed. See
|
|
913
|
+
* `TrainingRuleRunRefusalCode`.
|
|
898
914
|
*/
|
|
899
915
|
async runTrainingRule(id) {
|
|
900
916
|
const res = await this._http.fetchPost(`/api/loop/training-rules/${encodeURIComponent(id)}/run`, {});
|
|
@@ -980,6 +996,47 @@ export class Loop {
|
|
|
980
996
|
async cancelTrainingRun(id, params = {}) {
|
|
981
997
|
return this._http.fetchPost(`/api/loop/training-runs/${encodeURIComponent(id)}/cancel`, params);
|
|
982
998
|
}
|
|
999
|
+
// ── pipelines ─────────────────────────────────────────────────────────
|
|
1000
|
+
/**
|
|
1001
|
+
* Every training rule in the workspace, seen as the series of versions it
|
|
1002
|
+
* produced: which exist, which one serves (`champion_version`), how each did
|
|
1003
|
+
* against the champion of its day and on the standing benchmark, the run in
|
|
1004
|
+
* flight and which version it will be, and what the pipeline is waiting for.
|
|
1005
|
+
*
|
|
1006
|
+
* Read-only. Every decision stays on the route that owns it -- promote,
|
|
1007
|
+
* reject and roll back on the run, the version limit and `explore_recipes`
|
|
1008
|
+
* on the rule.
|
|
1009
|
+
*
|
|
1010
|
+
* RETURNED WITH ITS ENVELOPE, because the list is not always all of them:
|
|
1011
|
+
* the service returns the newest 100, `total` is every pipeline in the
|
|
1012
|
+
* workspace and `truncated` is true when more exist than were returned. A
|
|
1013
|
+
* caller that shows `pipelines` alone presents a short list as the whole of
|
|
1014
|
+
* it. The rest are reached through `listTrainingRules`, which pages.
|
|
1015
|
+
*/
|
|
1016
|
+
async listPipelines() {
|
|
1017
|
+
const res = await this._http.fetchGet('/api/loop/pipelines');
|
|
1018
|
+
const pipelines = res.pipelines || [];
|
|
1019
|
+
return {
|
|
1020
|
+
pipelines,
|
|
1021
|
+
// An older service sends neither: its list is taken at its word.
|
|
1022
|
+
total: typeof res.total === 'number' ? res.total : pipelines.length,
|
|
1023
|
+
truncated: res.truncated === true,
|
|
1024
|
+
};
|
|
1025
|
+
}
|
|
1026
|
+
/**
|
|
1027
|
+
* One pipeline, by its training rule's id.
|
|
1028
|
+
*
|
|
1029
|
+
* `status` says whether it is the platform working (`running`), the member
|
|
1030
|
+
* who has to act (`needs_review`, `needs_funds`), or nothing at all until
|
|
1031
|
+
* somebody does (`paused`, with `paused_reason` or `needs_consent` saying
|
|
1032
|
+
* why; `complete` at `max_versions`). `month_spent_cents` is the figure the
|
|
1033
|
+
* monthly limit is enforced against: what runs started this month have cost
|
|
1034
|
+
* at most, not an exact spend.
|
|
1035
|
+
*/
|
|
1036
|
+
async getPipeline(id) {
|
|
1037
|
+
const res = await this._http.fetchGet(`/api/loop/pipelines/${encodeURIComponent(id)}`);
|
|
1038
|
+
return res.pipeline;
|
|
1039
|
+
}
|
|
983
1040
|
// ── the comparison report ─────────────────────────────────────────────
|
|
984
1041
|
/**
|
|
985
1042
|
* The comparison behind a verdict: both models on the same held-out rows,
|
package/dist/types.d.ts
CHANGED
|
@@ -8,13 +8,13 @@ export interface BiOSConfig {
|
|
|
8
8
|
orgId?: string;
|
|
9
9
|
/** Workspace ID. Optional when using API keys (resolved from the key). Can override for multi-workspace keys. */
|
|
10
10
|
workspaceId?: string;
|
|
11
|
-
/** Base URL for the API. Falls back to the RUNBIOS_BASE_URL environment variable (legacy: BIOS_BASE_URL), then the canonical https://api.runbios.ai hostname. */
|
|
11
|
+
/** Base URL for the API. Falls back to the RUNBIOS_BASE_URL environment variable (legacy: BIOS_BASE_URL), then the canonical https://api-dev.runbios.ai hostname. */
|
|
12
12
|
baseUrl?: string;
|
|
13
13
|
/** Request timeout in milliseconds. Defaults to 30000. */
|
|
14
14
|
timeout?: number;
|
|
15
15
|
/** Default per-deployment inference key. Can be overridden per inference call. */
|
|
16
16
|
inferenceKey?: string;
|
|
17
|
-
/** Inference base URL. Defaults to baseUrl, then https://api.runbios.ai. */
|
|
17
|
+
/** Inference base URL. Defaults to baseUrl, then https://api-dev.runbios.ai. */
|
|
18
18
|
inferenceBaseUrl?: string;
|
|
19
19
|
/** End-to-end inference timeout in milliseconds. Defaults to 15 minutes. */
|
|
20
20
|
inferenceTimeout?: number;
|
|
@@ -2533,6 +2533,28 @@ export interface LoopSignalParams {
|
|
|
2533
2533
|
reason?: string;
|
|
2534
2534
|
author?: string;
|
|
2535
2535
|
metadata?: Record<string, unknown>;
|
|
2536
|
+
/**
|
|
2537
|
+
* Which training pipeline this feedback feeds, named in the SAME call.
|
|
2538
|
+
*
|
|
2539
|
+
* A build rule selects conversations by label and a training rule trains from
|
|
2540
|
+
* that build rule, so a label IS the pipeline a conversation goes down.
|
|
2541
|
+
* Putting one on used to need a second request after this one — and that
|
|
2542
|
+
* second request is the one that gets skipped, by a script that handles the
|
|
2543
|
+
* 201 and moves on, by a retry that succeeds here and fails there, by an
|
|
2544
|
+
* integration whose author never knew labelling was a step.
|
|
2545
|
+
*
|
|
2546
|
+
* What it leaves behind is a reviewed conversation in no pipeline: counted in
|
|
2547
|
+
* every "reviewed" total and selected by nothing.
|
|
2548
|
+
*
|
|
2549
|
+
* Written in one transaction with the verdict: either both land or neither
|
|
2550
|
+
* does. Omit them and nothing changes — the overwhelming majority of feedback
|
|
2551
|
+
* names no pipeline and behaves exactly as it always has.
|
|
2552
|
+
*/
|
|
2553
|
+
labels?: string[];
|
|
2554
|
+
/** The same, by dimension: `{ category: 'billing' }`. */
|
|
2555
|
+
attributes?: Record<string, string>;
|
|
2556
|
+
/** The master group the bare labels belong to. */
|
|
2557
|
+
parent?: string;
|
|
2536
2558
|
}
|
|
2537
2559
|
export interface LoopTrace {
|
|
2538
2560
|
id: string;
|
|
@@ -3125,9 +3147,46 @@ export type TrainingRulePausedReason = 'consent_invalid' | 'insufficient_funds'
|
|
|
3125
3147
|
* republished to keep up with a server-side capability flag.
|
|
3126
3148
|
*/
|
|
3127
3149
|
export type LoopTrainingMethod = 'sft' | 'rlhf';
|
|
3150
|
+
/**
|
|
3151
|
+
* What a rule's stored `train_type` may READ as.
|
|
3152
|
+
*
|
|
3153
|
+
* Still three, and deliberately: the column has always allowed `full`, so a
|
|
3154
|
+
* rule created before standing rules were narrowed to adapters can still come
|
|
3155
|
+
* back carrying it. Narrowing the response type would make this SDK
|
|
3156
|
+
* misrepresent a row that really exists.
|
|
3157
|
+
*/
|
|
3128
3158
|
export type TrainType = 'lora' | 'qlora' | 'full';
|
|
3159
|
+
/**
|
|
3160
|
+
* What a rule may be SET to, which is narrower.
|
|
3161
|
+
*
|
|
3162
|
+
* loop-service refuses `full` on preflight, create and update for a standing
|
|
3163
|
+
* rule: a rule trains unattended and on repeat, and a full fine-tune rewrites
|
|
3164
|
+
* every weight instead of adding a small adapter, so it cannot be compared or
|
|
3165
|
+
* rolled back cheaply. Typing the request as the wider set handed callers a
|
|
3166
|
+
* value guaranteed to come back a 400. One-off full fine-tunes are unaffected
|
|
3167
|
+
* — they are a different API.
|
|
3168
|
+
*/
|
|
3169
|
+
export type RuleTrainType = 'lora' | 'qlora';
|
|
3129
3170
|
export type GradersScope = 'source' | 'all' | 'none';
|
|
3130
|
-
|
|
3171
|
+
/**
|
|
3172
|
+
* Why a run fired. `variant` is a recipe variant: the pipeline had nothing new
|
|
3173
|
+
* to learn, so it trained the same base model on the same conversations with
|
|
3174
|
+
* one training setting changed (see {@link TrainingRecipe}).
|
|
3175
|
+
*/
|
|
3176
|
+
export type TrainingTrigger = 'rows' | 'cadence' | 'both' | 'manual' | 'variant';
|
|
3177
|
+
/**
|
|
3178
|
+
* The four training settings a recipe variant may change, as the run was
|
|
3179
|
+
* actually trained: the training service's own defaults are filled in where
|
|
3180
|
+
* the rule left a knob unset, so two recipes can be compared value for value.
|
|
3181
|
+
* A knob that does not apply to the run (a LoRA rank on a full fine-tune) is
|
|
3182
|
+
* null, never a guessed number.
|
|
3183
|
+
*/
|
|
3184
|
+
export interface TrainingRecipe {
|
|
3185
|
+
learning_rate: number | null;
|
|
3186
|
+
num_train_epochs: number | null;
|
|
3187
|
+
lora_rank: number | null;
|
|
3188
|
+
lora_alpha: number | null;
|
|
3189
|
+
}
|
|
3131
3190
|
export type TrainingRunState = 'built' | 'registering' | 'dataset_validating' | 'preflighting' | 'job_creating' | 'training' | 'checkpoint_ready' | 'candidate_booking' | 'candidate_running' | 'evaluating' | 'reported' | 'awaiting_review' | 'promoting' | 'waiting_funds' | 'promoted' | 'rejected' | 'expired' | 'failed' | 'budget_stopped' | 'cancelled' | 'superseded' | 'rolled_back';
|
|
3132
3191
|
/** The closed set a run never leaves. */
|
|
3133
3192
|
export declare const TERMINAL_RUN_STATES: readonly TrainingRunState[];
|
|
@@ -3218,6 +3277,30 @@ export interface TrainingRule {
|
|
|
3218
3277
|
eval_max_rows: number;
|
|
3219
3278
|
eval_max_tokens: number;
|
|
3220
3279
|
min_holdout_rows: number;
|
|
3280
|
+
/**
|
|
3281
|
+
* How many versions this pipeline may make, 1 to 10 (five unless changed).
|
|
3282
|
+
* A version is a trained model whose comparison was reported having scored
|
|
3283
|
+
* at least one conversation, or one a member put live; a run that failed,
|
|
3284
|
+
* was cancelled or compared nothing is not one and does not count. At the
|
|
3285
|
+
* limit the pipeline stops, and a manual run is refused with
|
|
3286
|
+
* `409 VERSION_LIMIT_REACHED`, until the limit is raised.
|
|
3287
|
+
*/
|
|
3288
|
+
max_versions: number;
|
|
3289
|
+
/**
|
|
3290
|
+
* Try other training recipes when there is nothing new to learn.
|
|
3291
|
+
*
|
|
3292
|
+
* MONEY-BEARING. When on, a pipeline that has made at least one version, and
|
|
3293
|
+
* has nothing reviewed since it that adds anything new to train on (nothing
|
|
3294
|
+
* reviewed at all, or only reviews -- thumbs-down with no correction, say --
|
|
3295
|
+
* that give it no row the last version's set lacked), trains the same base
|
|
3296
|
+
* model on the same conversations with ONE setting changed, and that model
|
|
3297
|
+
* takes over only if it beats what serves the app, like any other version.
|
|
3298
|
+
* Each try is a full paid run inside the rule's per-run ceilings and its
|
|
3299
|
+
* monthly limit, so changing this in EITHER direction changes the terms: the
|
|
3300
|
+
* rule stops firing until the member accepts them again. False unless
|
|
3301
|
+
* somebody turned it on.
|
|
3302
|
+
*/
|
|
3303
|
+
explore_recipes: boolean;
|
|
3221
3304
|
/**
|
|
3222
3305
|
* The standing benchmark replayed on every run of this rule, beside the
|
|
3223
3306
|
* per-run comparison and never instead of it. Null is the ordinary state.
|
|
@@ -3227,6 +3310,32 @@ export interface TrainingRule {
|
|
|
3227
3310
|
* its own. It raises no amount, so it does not invalidate consent.
|
|
3228
3311
|
*/
|
|
3229
3312
|
benchmark_id: string | null;
|
|
3313
|
+
/**
|
|
3314
|
+
* Whether that benchmark DECIDES the verdict, or only reports a number.
|
|
3315
|
+
*
|
|
3316
|
+
* False for every rule that has not asked, which is the point of it being
|
|
3317
|
+
* separate from `benchmark_id`. Attaching a benchmark means "replay this
|
|
3318
|
+
* fixed set on every run and put the score on the report"; it does not mean
|
|
3319
|
+
* "let that score overrule the comparison this rule was built on".
|
|
3320
|
+
*
|
|
3321
|
+
* When it is on it cuts both ways. A run whose held-out split came up short
|
|
3322
|
+
* can be decided at all -- before this, such a run trained a model, billed a
|
|
3323
|
+
* GPU and could never promote -- and a candidate that wins on fresh
|
|
3324
|
+
* conversations while losing ground on the fixed set is refused.
|
|
3325
|
+
*/
|
|
3326
|
+
benchmark_decides: boolean;
|
|
3327
|
+
/**
|
|
3328
|
+
* How many of the benchmark's pinned conversations must have scored before
|
|
3329
|
+
* it may decide anything. A replay that got through four of its forty has
|
|
3330
|
+
* not measured the new model.
|
|
3331
|
+
*/
|
|
3332
|
+
benchmark_min_rows: number;
|
|
3333
|
+
/**
|
|
3334
|
+
* The bar on the replay's own candidate-minus-incumbent delta. Like-for-like
|
|
3335
|
+
* WITHIN one replay -- both models, same pinned rows, same frozen judge, one
|
|
3336
|
+
* pass -- and not comparable between runs.
|
|
3337
|
+
*/
|
|
3338
|
+
benchmark_min_delta: number;
|
|
3230
3339
|
auto_promote: boolean;
|
|
3231
3340
|
promote_margin: number;
|
|
3232
3341
|
promote_min_win_rate: number;
|
|
@@ -3307,6 +3416,14 @@ export interface TrainingRulePreflight {
|
|
|
3307
3416
|
key_check: TrainingRuleKeyCheck;
|
|
3308
3417
|
terms_text: string;
|
|
3309
3418
|
terms_version: string;
|
|
3419
|
+
/**
|
|
3420
|
+
* The machines the hourly amounts are per hour of: the ladders the platform
|
|
3421
|
+
* chose when the request left them empty, or the caller's own echoed back
|
|
3422
|
+
* unchanged. An amount per hour means nothing without knowing what it is per
|
|
3423
|
+
* hour of, so show these beside the estimate before anyone accepts it.
|
|
3424
|
+
*/
|
|
3425
|
+
train_gpu_priorities: TrainingGPURung[];
|
|
3426
|
+
deploy_gpu_priorities: TrainingGPURung[];
|
|
3310
3427
|
}
|
|
3311
3428
|
export interface TrainingRuleBuildSpec {
|
|
3312
3429
|
method: string;
|
|
@@ -3340,6 +3457,18 @@ export interface TrainingRuleTriggerInput {
|
|
|
3340
3457
|
combinator?: TrainingCombinator;
|
|
3341
3458
|
/** null removes the row floor, leaving the schedule as the only trigger. */
|
|
3342
3459
|
min_new_rows?: number | null;
|
|
3460
|
+
/** 1 to 10. Absent leaves it as it is; absent on a create means five. */
|
|
3461
|
+
max_versions?: number;
|
|
3462
|
+
/**
|
|
3463
|
+
* See {@link TrainingRule.explore_recipes}: every variant it allows is a
|
|
3464
|
+
* full paid run, so changing it -- on OR off -- is a money-bearing edit that
|
|
3465
|
+
* pauses the rule, and it makes no version of any kind, until the new terms
|
|
3466
|
+
* are accepted. Turning it off to save money stops the pipeline too, until
|
|
3467
|
+
* somebody accepts again. It sits beside `max_versions`
|
|
3468
|
+
* because both say when the pipeline makes another version. Absent leaves
|
|
3469
|
+
* it as it is; absent on a create means off.
|
|
3470
|
+
*/
|
|
3471
|
+
explore_recipes?: boolean;
|
|
3343
3472
|
}
|
|
3344
3473
|
export interface TrainingRuleTrainingInput {
|
|
3345
3474
|
model_id?: string;
|
|
@@ -3351,7 +3480,7 @@ export interface TrainingRuleTrainingInput {
|
|
|
3351
3480
|
* plain supervised training means.
|
|
3352
3481
|
*/
|
|
3353
3482
|
rlhf_type?: string | null;
|
|
3354
|
-
train_type?:
|
|
3483
|
+
train_type?: RuleTrainType;
|
|
3355
3484
|
config?: Record<string, unknown>;
|
|
3356
3485
|
train_gpu_priorities?: TrainingGPURung[];
|
|
3357
3486
|
train_max_price_hour_cents?: number;
|
|
@@ -3434,6 +3563,19 @@ export interface TrainingRuleCreateRequest extends TrainingRulePreflightRequest
|
|
|
3434
3563
|
* preflight body and not on the update body.
|
|
3435
3564
|
*/
|
|
3436
3565
|
benchmark_id?: string;
|
|
3566
|
+
/**
|
|
3567
|
+
* And whether that benchmark decides, for the same reason `benchmark_id` is
|
|
3568
|
+
* here: run 1 reads its gate off the snapshot it freezes, and a gate applied
|
|
3569
|
+
* by a second call lands after run 1 has been decided without it.
|
|
3570
|
+
*
|
|
3571
|
+
* `400 BENCHMARK_GATE_HAS_NO_SET` when it is true with no benchmark,
|
|
3572
|
+
* `400 BENCHMARK_GATE_NEEDS_MIN_ROWS` when it is true with no floor.
|
|
3573
|
+
*/
|
|
3574
|
+
benchmark_decides?: boolean;
|
|
3575
|
+
/** Cannot exceed the set's size: `400 BENCHMARK_GATE_UNREACHABLE`. */
|
|
3576
|
+
benchmark_min_rows?: number;
|
|
3577
|
+
/** Zero -- the default -- means "must not lose ground". */
|
|
3578
|
+
benchmark_min_delta?: number;
|
|
3437
3579
|
}
|
|
3438
3580
|
export interface TrainingRuleUpdateRequest extends TrainingRulePreflightRequest {
|
|
3439
3581
|
expected_revision?: number;
|
|
@@ -3479,6 +3621,18 @@ export interface TrainingRun {
|
|
|
3479
3621
|
authorized_by: string;
|
|
3480
3622
|
rule_snapshot: Record<string, unknown>;
|
|
3481
3623
|
trigger: TrainingTrigger;
|
|
3624
|
+
/**
|
|
3625
|
+
* For a recipe variant, what it changed, in plain words ("half the learning
|
|
3626
|
+
* rate"). Null for every run that trained on the ordinary recipe.
|
|
3627
|
+
*/
|
|
3628
|
+
recipe_note: string | null;
|
|
3629
|
+
/**
|
|
3630
|
+
* The version whose recipe this run varied. Null when the variant started
|
|
3631
|
+
* from the rule's own settings, and for every run that is not a variant.
|
|
3632
|
+
*/
|
|
3633
|
+
recipe_of_version: number | null;
|
|
3634
|
+
/** The four recipe knobs this run trained with, defaults filled in. */
|
|
3635
|
+
recipe: TrainingRecipe;
|
|
3482
3636
|
fired_reason: string | null;
|
|
3483
3637
|
state: TrainingRunState;
|
|
3484
3638
|
state_entered_at: string;
|
|
@@ -3487,6 +3641,13 @@ export interface TrainingRun {
|
|
|
3487
3641
|
last_reason: string | null;
|
|
3488
3642
|
error_code: string | null;
|
|
3489
3643
|
last_error: string | null;
|
|
3644
|
+
/**
|
|
3645
|
+
* What to DO about the failure, in one sentence, derived server-side from
|
|
3646
|
+
* `error_code` -- the same words the failure email carries, so the two
|
|
3647
|
+
* cannot drift. Null when the run has not failed, or when the failure has
|
|
3648
|
+
* no remedy worth printing.
|
|
3649
|
+
*/
|
|
3650
|
+
error_next_step: string | null;
|
|
3490
3651
|
training_ceiling_cents: number;
|
|
3491
3652
|
candidate_ceiling_cents: number;
|
|
3492
3653
|
eval_ceiling_cents: number;
|
|
@@ -3503,6 +3664,13 @@ export interface TrainingRun {
|
|
|
3503
3664
|
queue_deadline_at: string | null;
|
|
3504
3665
|
checkpoint_id: string | null;
|
|
3505
3666
|
checkpoint_step: number | null;
|
|
3667
|
+
/**
|
|
3668
|
+
* Which version of the pipeline this run produced. Null for a run that is
|
|
3669
|
+
* not a version: one that failed, was cancelled, or whose comparison scored
|
|
3670
|
+
* nothing, and that nobody put live. `seq` counts attempts; this counts
|
|
3671
|
+
* models that competed, plus any a member promoted without a comparison.
|
|
3672
|
+
*/
|
|
3673
|
+
version_no: number | null;
|
|
3506
3674
|
training_eval_loss: number | null;
|
|
3507
3675
|
candidate_deployment_id: string | null;
|
|
3508
3676
|
candidate_name: string | null;
|
|
@@ -3546,10 +3714,20 @@ export interface TrainingRunSummary {
|
|
|
3546
3714
|
workspace_id: string;
|
|
3547
3715
|
seq: number;
|
|
3548
3716
|
trigger: TrainingTrigger;
|
|
3717
|
+
/** See TrainingRun.recipe_note. On the list row too: the Runs table is where a variant is first seen. */
|
|
3718
|
+
recipe_note: string | null;
|
|
3719
|
+
/** See TrainingRun.recipe_of_version. */
|
|
3720
|
+
recipe_of_version: number | null;
|
|
3721
|
+
/** See TrainingRun.recipe. */
|
|
3722
|
+
recipe: TrainingRecipe;
|
|
3549
3723
|
state: TrainingRunState;
|
|
3550
3724
|
state_entered_at: string;
|
|
3551
3725
|
last_reason: string | null;
|
|
3552
3726
|
error_code: string | null;
|
|
3727
|
+
/** See TrainingRun.error_next_step. On the list row too: the Runs table is where a failure is first seen. */
|
|
3728
|
+
error_next_step: string | null;
|
|
3729
|
+
/** See TrainingRun.version_no. */
|
|
3730
|
+
version_no: number | null;
|
|
3553
3731
|
verdict: TrainingVerdict | null;
|
|
3554
3732
|
decision: TrainingDecision | null;
|
|
3555
3733
|
train_rows: number | null;
|
|
@@ -3975,13 +4153,290 @@ export interface BenchmarkRetireParams {
|
|
|
3975
4153
|
reason?: string;
|
|
3976
4154
|
}
|
|
3977
4155
|
/**
|
|
3978
|
-
* Attach a benchmark to a rule,
|
|
4156
|
+
* Attach a benchmark to a rule, detach it with a present null, or move the bar
|
|
4157
|
+
* it decides on.
|
|
4158
|
+
*
|
|
4159
|
+
* ABSENT IS "LEAVE IT", PRESENT IS "MAKE IT THIS", and that applies to every
|
|
4160
|
+
* field here. This is the route that attaches a benchmark AND the route that
|
|
4161
|
+
* moves its bar a month later, so a call naming only `benchmark_id` must not
|
|
4162
|
+
* reset a floor somebody chose, and a call naming only `benchmark_min_delta`
|
|
4163
|
+
* must not detach the set.
|
|
3979
4164
|
*
|
|
3980
|
-
*
|
|
3981
|
-
*
|
|
4165
|
+
* Detaching is the one exception: a present null `benchmark_id` clears
|
|
4166
|
+
* `benchmark_decides` with it, because a gate with nothing behind it is a
|
|
4167
|
+
* verdict waiting on a measurement that will never come. Asking for both in
|
|
4168
|
+
* one request -- null id and `benchmark_decides: true` -- is refused rather
|
|
4169
|
+
* than resolved, because either resolution is a guess about which half you
|
|
4170
|
+
* meant.
|
|
3982
4171
|
*/
|
|
3983
4172
|
export interface TrainingRuleBenchmarkRequest {
|
|
4173
|
+
benchmark_id?: string | null;
|
|
4174
|
+
benchmark_decides?: boolean;
|
|
4175
|
+
benchmark_min_rows?: number;
|
|
4176
|
+
benchmark_min_delta?: number;
|
|
4177
|
+
}
|
|
4178
|
+
/**
|
|
4179
|
+
* The machine codes `runTrainingRule` can be refused with. Each is the
|
|
4180
|
+
* platform declining to start a paid run it would only refuse a minute later,
|
|
4181
|
+
* so each is said in front of the caller rather than left for the rule's
|
|
4182
|
+
* `last_reason` to explain after they have stopped looking.
|
|
4183
|
+
*
|
|
4184
|
+
* - `CONSENT_REQUIRED` (409): the terms changed and nobody accepted them.
|
|
4185
|
+
* - `RULE_PAUSED` (409): the rule is switched off or the platform stopped it.
|
|
4186
|
+
* - `RUN_ACTIVE` (409): one run at a time, so two cannot race to promote.
|
|
4187
|
+
* - `VERSION_LIMIT_REACHED` (409): the pipeline has made `max_versions`
|
|
4188
|
+
* versions. Raise the limit or start a new pipeline.
|
|
4189
|
+
* - `MONTHLY_LIMIT_REACHED` (409): the most this month's runs can have cost
|
|
4190
|
+
* plus the most one more run may cost would pass `monthly_ceiling_cents`.
|
|
4191
|
+
* The body carries the figures:
|
|
4192
|
+
* see {@link TrainingRuleMonthlyLimitRefusal}.
|
|
4193
|
+
* - `AGENT_OFF` (409): the workspace's Conscious Loop agent is not turned on.
|
|
4194
|
+
* Every comparison runs under the agent's key, so a run started without one
|
|
4195
|
+
* would pay for training and could not be compared. Turn it on in Agent
|
|
4196
|
+
* settings, then run the rule.
|
|
4197
|
+
* - `NOT_ENOUGH_ROWS` (422): too few reviewed or held-out conversations, with
|
|
4198
|
+
* `train_rows`, `holdout_rows` and the `floor` it needed.
|
|
4199
|
+
*/
|
|
4200
|
+
export type TrainingRuleRunRefusalCode = 'CONSENT_REQUIRED' | 'RULE_PAUSED' | 'RUN_ACTIVE' | 'VERSION_LIMIT_REACHED' | 'MONTHLY_LIMIT_REACHED' | 'AGENT_OFF' | 'NOT_ENOUGH_ROWS';
|
|
4201
|
+
/**
|
|
4202
|
+
* The machine codes `deleteDataset` can be refused with.
|
|
4203
|
+
*
|
|
4204
|
+
* - `DATASET_IN_USE` (409): a training run was trained on this set, so it is
|
|
4205
|
+
* kept as the record of what that run learned from.
|
|
4206
|
+
*/
|
|
4207
|
+
export type LoopDatasetDeleteRefusalCode = 'DATASET_IN_USE';
|
|
4208
|
+
/**
|
|
4209
|
+
* The body of a manual run refused `409 MONTHLY_LIMIT_REACHED`, as
|
|
4210
|
+
* `ApiError.body` carries it.
|
|
4211
|
+
*
|
|
4212
|
+
* The test is the worst case, not the average: a run may start only if the
|
|
4213
|
+
* most the rule's runs this UTC calendar month can have cost
|
|
4214
|
+
* (`month_spent_cents`), plus the most the run about to start may spend (its
|
|
4215
|
+
* three ceilings added up), fits under the limit.
|
|
4216
|
+
*/
|
|
4217
|
+
export interface TrainingRuleMonthlyLimitRefusal {
|
|
4218
|
+
error: {
|
|
4219
|
+
code: 'MONTHLY_LIMIT_REACHED';
|
|
4220
|
+
message: string;
|
|
4221
|
+
};
|
|
4222
|
+
/** What this rule's runs have cost at most this UTC calendar month; see {@link Pipeline.month_spent_cents}. */
|
|
4223
|
+
month_spent_cents: number;
|
|
4224
|
+
/** The rule's monthly limit. */
|
|
4225
|
+
monthly_ceiling_cents: number;
|
|
4226
|
+
/** The most one run of this rule may spend: training + candidate + evaluation ceilings. */
|
|
4227
|
+
run_max_cents: number;
|
|
4228
|
+
/**
|
|
4229
|
+
* The first instant of the next UTC month, when the counted spend starts
|
|
4230
|
+
* again from nothing. When `run_max_cents` alone exceeds the limit, no month
|
|
4231
|
+
* will ever fit and only raising the limit helps; `message` says which.
|
|
4232
|
+
*/
|
|
4233
|
+
resumes_at: string;
|
|
4234
|
+
}
|
|
4235
|
+
/** The most versions a pipeline may be set to make (loop-service maxVersionsCeiling). */
|
|
4236
|
+
export declare const PIPELINE_MAX_VERSIONS_CEILING = 10;
|
|
4237
|
+
/**
|
|
4238
|
+
* What a pipeline is doing. `needs_review` and `needs_funds` are a finished
|
|
4239
|
+
* version waiting on the MEMBER -- a decision, or a top-up -- which is not the
|
|
4240
|
+
* platform working on it.
|
|
4241
|
+
*
|
|
4242
|
+
* `paused` is a rule that is switched off, AND an enabled rule the platform
|
|
4243
|
+
* has stopped firing (`paused_reason`) or whose terms changed and were never
|
|
4244
|
+
* accepted (`needs_consent`). `complete` is a pipeline at its `max_versions`.
|
|
4245
|
+
*/
|
|
4246
|
+
export type PipelineStatus = 'running' | 'needs_review' | 'needs_funds' | 'waiting' | 'paused' | 'complete';
|
|
4247
|
+
/**
|
|
4248
|
+
* The run in flight. It is not a version until its comparison is reported
|
|
4249
|
+
* having scored at least one conversation, or a member puts it live.
|
|
4250
|
+
*/
|
|
4251
|
+
export interface PipelineActiveRun {
|
|
4252
|
+
run_id: string;
|
|
4253
|
+
state: TrainingRunState;
|
|
4254
|
+
since: string;
|
|
4255
|
+
/**
|
|
4256
|
+
* The version number it will take if it becomes one: if its comparison
|
|
4257
|
+
* scores something, or if a member promotes it anyway.
|
|
4258
|
+
*/
|
|
4259
|
+
will_be_version: number;
|
|
4260
|
+
version_no: number | null;
|
|
4261
|
+
last_reason: string | null;
|
|
4262
|
+
last_error: string | null;
|
|
4263
|
+
train_rows: number | null;
|
|
4264
|
+
holdout_rows: number | null;
|
|
4265
|
+
/**
|
|
4266
|
+
* Its comparison scored at least one conversation. False on a reported run
|
|
4267
|
+
* means the held-back set was too small to compare anything: that run is
|
|
4268
|
+
* not a version, and becomes one only if a member promotes it.
|
|
4269
|
+
*/
|
|
4270
|
+
competed: boolean;
|
|
4271
|
+
/** What recipe it is trying, when it is a recipe variant. See TrainingRun.recipe_note. */
|
|
4272
|
+
recipe_note: string | null;
|
|
4273
|
+
}
|
|
4274
|
+
/**
|
|
4275
|
+
* One version: a trained model whose comparison scored at least one
|
|
4276
|
+
* conversation, or one a member put live without one. The second kind has
|
|
4277
|
+
* `rows_scored` 0 and `win_rate` null, and may be the champion.
|
|
4278
|
+
*/
|
|
4279
|
+
export interface PipelineVersion {
|
|
4280
|
+
version: number;
|
|
4281
|
+
run_id: string;
|
|
4282
|
+
seq: number;
|
|
4283
|
+
state: TrainingRunState;
|
|
4284
|
+
verdict: TrainingVerdict | null;
|
|
4285
|
+
decision: TrainingDecision | null;
|
|
4286
|
+
decided_at: string | null;
|
|
4287
|
+
/** True for the version the app is served by today. */
|
|
4288
|
+
is_champion: boolean;
|
|
4289
|
+
checkpoint_id: string | null;
|
|
4290
|
+
train_rows: number | null;
|
|
4291
|
+
holdout_rows: number | null;
|
|
4292
|
+
/** Head-to-head against the champion of its day, on conversations neither trained on. */
|
|
4293
|
+
rows_scored: number | null;
|
|
4294
|
+
win_rate: number | null;
|
|
4295
|
+
mean_delta: number | null;
|
|
4296
|
+
/**
|
|
4297
|
+
* The standing benchmark's replay of this version, when the rule has one.
|
|
4298
|
+
* Only scores with `benchmark_comparable` sit on one axis.
|
|
4299
|
+
*/
|
|
4300
|
+
benchmark_score: number | null;
|
|
4301
|
+
benchmark_rows: number | null;
|
|
4302
|
+
benchmark_rows_total: number | null;
|
|
4303
|
+
benchmark_status: string | null;
|
|
4304
|
+
/**
|
|
4305
|
+
* False for a score that is not on today's yardstick: measured on an earlier
|
|
4306
|
+
* revision of the benchmark's items, or on a different benchmark than the
|
|
4307
|
+
* one attached now. Never draw or subtract those on the same line.
|
|
4308
|
+
*/
|
|
4309
|
+
benchmark_comparable: boolean;
|
|
4310
|
+
/**
|
|
4311
|
+
* Everything this version's run cost, in cents: training, the comparison
|
|
4312
|
+
* machine, the judge and the benchmark replay.
|
|
4313
|
+
*/
|
|
4314
|
+
cost_cents: number;
|
|
4315
|
+
/**
|
|
4316
|
+
* True when no charge was recorded for the comparison machine (a run made
|
|
4317
|
+
* before the platform recorded one) and `cost_cents` counts it at its
|
|
4318
|
+
* ceiling instead, so the figure is the most the version can have cost,
|
|
4319
|
+
* not what it did.
|
|
4320
|
+
*/
|
|
4321
|
+
cost_includes_candidate_ceiling: boolean;
|
|
4322
|
+
created_at: string;
|
|
4323
|
+
/**
|
|
4324
|
+
* When this version's head-to-head started. The champion it faced is the
|
|
4325
|
+
* model that was serving at that moment, which is why a delta against "the
|
|
4326
|
+
* champion before it" is decided by this and not by `created_at`. Null when
|
|
4327
|
+
* no comparison was ever started for it.
|
|
4328
|
+
*/
|
|
4329
|
+
compared_at: string | null;
|
|
4330
|
+
/**
|
|
4331
|
+
* Why the run that made it fired. `variant` is a version trained on the same
|
|
4332
|
+
* conversations as the one before it with one setting changed; every other
|
|
4333
|
+
* value is a version that learned from newly reviewed conversations (or a
|
|
4334
|
+
* member's run now).
|
|
4335
|
+
*/
|
|
4336
|
+
trigger: TrainingTrigger;
|
|
4337
|
+
/** See TrainingRun.recipe_note. */
|
|
4338
|
+
recipe_note: string | null;
|
|
4339
|
+
/** See TrainingRun.recipe_of_version. */
|
|
4340
|
+
recipe_of_version: number | null;
|
|
4341
|
+
/** See TrainingRun.recipe. */
|
|
4342
|
+
recipe: TrainingRecipe;
|
|
4343
|
+
finished_at: string | null;
|
|
4344
|
+
}
|
|
4345
|
+
/**
|
|
4346
|
+
* A training rule seen as the series of versions it produces: which exist,
|
|
4347
|
+
* which one serves, whether each beat the one before it, and what happens
|
|
4348
|
+
* next. READ-ONLY: every decision stays on the route that owns it (promote,
|
|
4349
|
+
* reject and roll back on the run; the version limit on the rule).
|
|
4350
|
+
*/
|
|
4351
|
+
/**
|
|
4352
|
+
* Whether a pipeline can make a recipe variant next, and when it cannot, which
|
|
4353
|
+
* reason -- each is a different next step.
|
|
4354
|
+
*
|
|
4355
|
+
* - `off`: explore_recipes is off.
|
|
4356
|
+
* - `waiting_for_v1`: on, but a variant varies a version and there is none yet.
|
|
4357
|
+
* - `available`: on, with `recipe_variants_left` untried variants that fit.
|
|
4358
|
+
* - `no_slots`: untried variants fit (`recipe_variants_left` > 0), but every
|
|
4359
|
+
* version slot is made or taken by the run in flight. Raise `max_versions`
|
|
4360
|
+
* (or, at 10, start a new pipeline) and one can run. Only when variants are
|
|
4361
|
+
* left: a service that reports `no_slots` with `recipe_variants_left` of 0
|
|
4362
|
+
* means `all_tried`, and raising the limit starts nothing.
|
|
4363
|
+
* - `all_tried`: every variant that fits has been tried on the conversations
|
|
4364
|
+
* it has now, whether or not a version slot is free. Raising `max_versions`
|
|
4365
|
+
* does not start one; new reviews bring the next version.
|
|
4366
|
+
* - `none_fit`: no variant fits this rule's training settings.
|
|
4367
|
+
*/
|
|
4368
|
+
export type PipelineRecipeExploration = 'off' | 'waiting_for_v1' | 'available' | 'no_slots' | 'all_tried' | 'none_fit';
|
|
4369
|
+
export interface Pipeline {
|
|
4370
|
+
rule_id: string;
|
|
4371
|
+
rule_name: string;
|
|
4372
|
+
enabled: boolean;
|
|
4373
|
+
/** The limit the member set, 1 to 10. */
|
|
4374
|
+
max_versions: number;
|
|
4375
|
+
/** Versions made so far. At `max_versions` the pipeline stops. */
|
|
4376
|
+
versions_made: number;
|
|
4377
|
+
/** The pinned base every version is a fresh adapter over, which is what makes their scores comparable. */
|
|
4378
|
+
base_model_id: string;
|
|
4379
|
+
base_model_revision: string;
|
|
4380
|
+
train_type: string;
|
|
4381
|
+
/** Null when what serves is not a version of this pipeline. */
|
|
4382
|
+
champion_version: number | null;
|
|
4383
|
+
/** The name the app calls. It does not change when a version is promoted. */
|
|
4384
|
+
serving_name: string;
|
|
4385
|
+
/**
|
|
4386
|
+
* Another pipeline that has since promoted onto the same serving name. Only
|
|
4387
|
+
* the most recent promotion serves; this pipeline's own champion no longer
|
|
4388
|
+
* does. Null when nothing has taken the name over.
|
|
4389
|
+
*/
|
|
4390
|
+
serving_taken_over_by: string | null;
|
|
3984
4391
|
benchmark_id: string | null;
|
|
4392
|
+
benchmark_name: string | null;
|
|
4393
|
+
benchmark_decides: boolean;
|
|
4394
|
+
benchmark_min_delta: number;
|
|
4395
|
+
status: PipelineStatus;
|
|
4396
|
+
/**
|
|
4397
|
+
* The rule's own sentence about what it is waiting for or why it stopped,
|
|
4398
|
+
* verbatim. Only as fresh as the platform's last visit: when `status` is
|
|
4399
|
+
* `paused`, `paused_reason` and `needs_consent` are what say why.
|
|
4400
|
+
*/
|
|
4401
|
+
next_reason: string | null;
|
|
4402
|
+
last_checked_at: string | null;
|
|
4403
|
+
next_due_at: string | null;
|
|
4404
|
+
auto_promote: boolean;
|
|
4405
|
+
/** Why the platform stopped firing this rule, or null. Same codes as the rule's. */
|
|
4406
|
+
paused_reason: TrainingRulePausedReason | null;
|
|
4407
|
+
/**
|
|
4408
|
+
* The rule's terms changed and nobody has accepted them. It makes nothing
|
|
4409
|
+
* until somebody does.
|
|
4410
|
+
*/
|
|
4411
|
+
needs_consent: boolean;
|
|
4412
|
+
/** The rule's explore_recipes, as it stands. */
|
|
4413
|
+
explore_recipes: boolean;
|
|
4414
|
+
/**
|
|
4415
|
+
* How many recipe variants that fit this rule's settings have not yet been
|
|
4416
|
+
* tried on the conversations the pipeline has now. Not capped by the free
|
|
4417
|
+
* version slots; 0 while exploration is off or when no variant fits. A count
|
|
4418
|
+
* is not a reason: read `recipe_exploration` for whether one can run next.
|
|
4419
|
+
*/
|
|
4420
|
+
recipe_variants_left: number;
|
|
4421
|
+
/** Whether the next recipe variant can be tried, and if not, why. */
|
|
4422
|
+
recipe_exploration: PipelineRecipeExploration;
|
|
4423
|
+
/** The rule's monthly limit. Null means it has none. */
|
|
4424
|
+
monthly_ceiling_cents: number | null;
|
|
4425
|
+
/**
|
|
4426
|
+
* What this rule's runs have cost at most, counted toward the monthly limit
|
|
4427
|
+
* this UTC calendar month -- the figure the limit is enforced against. An
|
|
4428
|
+
* upper bound, not an exact spend. A run counts toward the month it was
|
|
4429
|
+
* created in, and a run still in progress also counts toward the current
|
|
4430
|
+
* month, at its three ceilings or what it has been billed when that is
|
|
4431
|
+
* more. A finished run counts what it was billed, and its comparison
|
|
4432
|
+
* machine is billed at the most it could have cost (the hourly cap for the
|
|
4433
|
+
* time it was up, never more than the candidate ceiling). A run created in
|
|
4434
|
+
* an earlier month that finishes in this one counts here only while it is
|
|
4435
|
+
* still running.
|
|
4436
|
+
*/
|
|
4437
|
+
month_spent_cents: number;
|
|
4438
|
+
active_run: PipelineActiveRun | null;
|
|
4439
|
+
versions: PipelineVersion[];
|
|
3985
4440
|
}
|
|
3986
4441
|
export interface TrainingRuleListResponse {
|
|
3987
4442
|
rules: TrainingRule[];
|
|
@@ -3990,6 +4445,7 @@ export interface TrainingRuleListResponse {
|
|
|
3990
4445
|
export interface TrainingRuleResponse {
|
|
3991
4446
|
rule: TrainingRule;
|
|
3992
4447
|
recent_runs: TrainingRunSummary[];
|
|
4448
|
+
/** What this rule's runs have cost at most this UTC calendar month; see {@link Pipeline.month_spent_cents}. */
|
|
3993
4449
|
month_spent_cents: number;
|
|
3994
4450
|
judge_agreement: JudgeAgreement | null;
|
|
3995
4451
|
}
|
|
@@ -4062,3 +4518,14 @@ export interface BenchmarkHistoryResponse {
|
|
|
4062
4518
|
points: BenchmarkHistoryPoint[];
|
|
4063
4519
|
total: number;
|
|
4064
4520
|
}
|
|
4521
|
+
export interface PipelineListResponse {
|
|
4522
|
+
/** The newest pipelines first, at most the service's page (100). */
|
|
4523
|
+
pipelines: Pipeline[];
|
|
4524
|
+
/** Every pipeline in the workspace, including any not returned. */
|
|
4525
|
+
total: number;
|
|
4526
|
+
/** More pipelines exist than were returned; the rest are reached through `listTrainingRules`. */
|
|
4527
|
+
truncated: boolean;
|
|
4528
|
+
}
|
|
4529
|
+
export interface PipelineResponse {
|
|
4530
|
+
pipeline: Pipeline;
|
|
4531
|
+
}
|
package/dist/types.js
CHANGED
|
@@ -12,3 +12,6 @@ export const TERMINAL_RUN_STATES = [
|
|
|
12
12
|
'superseded',
|
|
13
13
|
'rolled_back',
|
|
14
14
|
];
|
|
15
|
+
/* ── Pipelines ────────────────────────────────────────────────────────────── */
|
|
16
|
+
/** The most versions a pipeline may be set to make (loop-service maxVersionsCeiling). */
|
|
17
|
+
export const PIPELINE_MAX_VERSIONS_CEILING = 10;
|