runbios-sdk 0.2.13-dev.243 → 0.2.13-rc.239

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/README.md CHANGED
@@ -491,7 +491,7 @@ try {
491
491
  | `apiKey` | `RUNBIOS_API_KEY` env var (legacy `BIOS_API_KEY`) | Dashboard-issued API key (default `bios-`; custom and provider-shaped prefixes also work; legacy `usf-` keys remain valid) |
492
492
  | `accessToken` | — | JWT access token |
493
493
  | `orgId` | — | Organization ID (auto-resolved with API keys) |
494
- | `workspaceId` | — | Workspace ID (auto-resolved with workspace-bound API keys). An explicit ID must match the key's bound workspace; a workspace-less key can select a workspace it belongs to within its bound organization. |
494
+ | `workspaceId` | — | Workspace ID (auto-resolved with API keys) |
495
495
  | `baseUrl` | `RUNBIOS_BASE_URL` env var (legacy `BIOS_BASE_URL`), then `https://api.runbios.ai` | Canonical production hostname (release-gated; this documentation does not assert current availability). During prelaunch/dev, pass `https://api-dev.runbios.ai` explicitly. |
496
496
  | `timeout` | `30000` | Request timeout in ms |
497
497
  | `inferenceKey` | `RUNBIOS_INFERENCE_KEY` env var (legacy `BIOS_INFERENCE_KEY`), then `apiKey` | Key used by `client.inference` for `/v1` calls. A per-deployment `sk-bios-...` key, or the platform `apiKey` itself when it carries the serverless scope — you never pass the same key twice |
package/dist/client.js CHANGED
@@ -340,7 +340,7 @@ export class HttpClient {
340
340
  constructor(config) {
341
341
  // Default host stays api.runbios.ai for now; cutover to api.runbios.ai is
342
342
  // planned once its DNS exists.
343
- this.baseUrl = (config.baseUrl || envBaseUrl() || 'https://api-dev.runbios.ai').replace(/\/+$/, '');
343
+ this.baseUrl = (config.baseUrl || envBaseUrl() || 'https://api-staging.runbios.ai').replace(/\/+$/, '');
344
344
  this.apiKey = config.apiKey ?? envApiKey();
345
345
  this.accessToken = config.accessToken;
346
346
  this.orgId = config.orgId;
package/dist/index.d.ts CHANGED
@@ -36,7 +36,7 @@ import { Loop } from './resources/loop.js';
36
36
  * SDK version. Sent as part of the User-Agent header.
37
37
  * Must match package.json "version" -- enforced by a contract test.
38
38
  */
39
- export declare const VERSION = "0.2.13-dev.243";
39
+ export declare const VERSION = "0.2.13-rc.239";
40
40
  export declare class RunBiOS {
41
41
  /** Search models, fetch configs, check adapter compatibility. */
42
42
  readonly models: Models;
@@ -75,8 +75,6 @@ export { Training } from './resources/training.js';
75
75
  export { Wallet } from './resources/wallet.js';
76
76
  export { GPU, type GPURecommendation } from './resources/gpu.js';
77
77
  export { Inference, validateChatRequest, parseSSE, buildInferenceRequest, contextSizingBasis, type ContextSizingBasis, CONTEXT_DEFAULT_CEILING, CONTEXT_EDITABLE_FLOOR, type ChatMessage, type FunctionTool, type ChatCompletionParams, type ChatCompletionResponse, type ChatCompletionChunk, } from './resources/inference.js';
78
- export type { BiOSConfig, PaginatedResponse, ApiErrorBody, AvailableGpuAlternative, CapacityMinimumRequirement, InferenceBookingAccepted, Model, ModelSearchParams, ModelSearchResponse, ModelDetailResponse, ModelConfig, Dataset, DatasetListParams, DatasetListResponse, DatasetUploadParams, DatasetPreview, DatasetPreviewParams, DatasetImportHFParams, DatasetRegisterHFParams, DatasetHubSearchParams, DatasetHubPreviewParams, DatasetValidation, DatasetFormatVariant, DatasetFormatSpec, DatasetFormatSpecs, DatasetStorageUsage, TrainingMethod, RLHFAlgorithm, AdapterType, TrainingJobStatus, TrainingStatusFilter, TrainingCreateParams, TrainingListParams, TrainingListResponse, TrainingJob, TrainingMetrics, TrainingResourceMetrics, TrainingDeviceMetrics, MetricPoint, MetricGraphConfig, TrainingCheckpoint, TrainingLogs, TrainingLogEntry, TrainingStopResponse, TrainingResumeResponse, CanonicalTrainingRequest, TrainingPreflightDataset, TrainingPreflightWarning, TrainingPreflightResponse, TrainingCapabilityChoice, TrainingConfigFieldCapability, TrainingCapabilities, GPUChoice, WalletBalance, Transaction, TransactionListResponse, TransactionListParams, GPUInfo, GPUPricingResponse, GPUOptionsParams, TrainingRecommendParams, TrainingAdvisorVerdict, TrainingAdvisorRecommendation, TrainingAdvisorUserShape, TrainingAdvisorJustification, TrainingAdvisorBasis, GPUOption, GPUOptionSuggestion, GPUOptionsResponse, InferenceStatus, InferenceCreateParams, InferenceUpdateParams, InferenceDeployment, InferenceDeploymentSummary, InferenceCreateResponse, InferenceListResponse, InferenceUpdateResponse, InferenceLifecycleResponse, InferenceDeleteResponse, InferenceGPUOptionMarket, InferenceGPUOption, InferenceGPUAlternative, InferenceGPUOptionsParams, InferenceGPUOptionsResponse, InferenceCanonicalRequest, InferencePreflightWarning, InferencePreflightResponse, AdapterCompatibility, AdapterCompatibilityResponse, AdapterCompatibilityParams, SupportedArchitecture, ArchitectureScope, SupportedArchitecturesResponse, ApiKey, ApiKeyScope, ApiKeyIntrospection, Organization, OrgMember, OrgInvite, Workspace, Integration, IntegrationCreateParams, StorageUsage, StorageObject, PromptTokensDetails, ChatCompletionUsage, TrainingCadence, TrainingCombinator, ServingKind, TrainingRulePausedReason, LoopTrainingMethod, TrainType, GradersScope, TrainingTrigger, TrainingRecipe, TrainingRunState, TrainingVerdict, TrainingDecision, NotifyState, EvaluationStatus, EvaluationItemStatus, EvaluationWinner, EvaluationWarningCode, ConsentVia, TrainingToolCall, TrainingMessage, TrainingRuleServing, TrainingGPURung, TrainingRule, TrainingRuleConsent, TrainingRulePreflightRefusal, TrainingRuleKeyRef, TrainingRuleKeyCheck, TrainingRulePreflight, TrainingRuleBuildSpec, TrainingRuleTriggerInput, TrainingRuleTrainingInput, TrainingRuleDeployInput, TrainingRuleMoneyInput, TrainingRuleEvaluationInput, TrainingRulePromotionInput, TrainingRuleAcceptTerms, TrainingRulePreflightRequest, TrainingRuleCreateRequest, TrainingRuleUpdateRequest, TrainingRuleConsentRequest, TrainingRuleListParams, TrainingRunPromoteRequest, TrainingRunRejectRequest, TrainingRunRollbackRequest, TrainingRunCancelRequest, TrainingRunListParams, TrainingRun, TrainingRunSummary, TrainingRunEvent, TrainingRunLinks, TrainingRunActions, EvaluationRef, EvaluationDecoding, EvaluationJudgeDimension, EvaluationDimensionScore, EvaluationGraderScore, EvaluationWarning, EvaluationMargin, JudgeAgreementDimension, JudgeAgreement, Evaluation, EvaluationItemSide, EvaluationItem, EvaluationItemListParams, JudgeAgreementParams, AgentSettings, AgentSettingsRequest, BenchmarkStatus, BenchmarkSourceKind, BenchmarkRunStatus, BenchmarkScoring, BenchmarkDimensionScore, Benchmark, BenchmarkItem, BenchmarkRun, BenchmarkHistoryPoint, BenchmarkSource, BenchmarkCreateParams, BenchmarkListParams, BenchmarkItemListParams, BenchmarkHistoryParams, BenchmarkRetireParams, TrainingRuleBenchmarkRequest, InferenceAlias, InferenceAliasRequest, TrainingRuleListResponse, TrainingRuleResponse, TrainingRuleMutationResponse, TrainingRuleDeleteResponse, TrainingRunListResponse, TrainingRunResponse, TrainingRunActionResponse, EvaluationItemsResponse, AgentSettingsResponse, InferenceAliasListResponse, InferenceAliasResponse, InferenceAliasDeleteResponse, BenchmarkListResponse, BenchmarkItemsResponse, BenchmarkHistoryResponse, TrainingRuleRunRefusalCode, TrainingRuleMonthlyLimitRefusal, LoopDatasetDeleteRefusalCode, PipelineStatus, PipelineRecipeExploration, PipelineActiveRun, PipelineVersion, Pipeline, PipelineListResponse, PipelineResponse, } from './types.js';
78
+ export type { BiOSConfig, PaginatedResponse, ApiErrorBody, AvailableGpuAlternative, CapacityMinimumRequirement, InferenceBookingAccepted, Model, ModelSearchParams, ModelSearchResponse, ModelDetailResponse, ModelConfig, Dataset, DatasetListParams, DatasetListResponse, DatasetUploadParams, DatasetPreview, DatasetPreviewParams, DatasetImportHFParams, DatasetRegisterHFParams, DatasetHubSearchParams, DatasetHubPreviewParams, DatasetValidation, DatasetFormatVariant, DatasetFormatSpec, DatasetFormatSpecs, DatasetStorageUsage, TrainingMethod, RLHFAlgorithm, AdapterType, TrainingJobStatus, TrainingStatusFilter, TrainingCreateParams, TrainingListParams, TrainingListResponse, TrainingJob, TrainingMetrics, TrainingResourceMetrics, TrainingDeviceMetrics, MetricPoint, MetricGraphConfig, TrainingCheckpoint, TrainingLogs, TrainingLogEntry, TrainingStopResponse, TrainingResumeResponse, CanonicalTrainingRequest, TrainingPreflightDataset, TrainingPreflightWarning, TrainingPreflightResponse, TrainingCapabilityChoice, TrainingConfigFieldCapability, TrainingCapabilities, GPUChoice, WalletBalance, Transaction, TransactionListResponse, TransactionListParams, GPUInfo, GPUPricingResponse, GPUOptionsParams, TrainingRecommendParams, TrainingAdvisorVerdict, TrainingAdvisorRecommendation, TrainingAdvisorUserShape, TrainingAdvisorJustification, TrainingAdvisorBasis, GPUOption, GPUOptionSuggestion, GPUOptionsResponse, InferenceStatus, InferenceCreateParams, InferenceUpdateParams, InferenceDeployment, InferenceDeploymentSummary, InferenceCreateResponse, InferenceListResponse, InferenceUpdateResponse, InferenceLifecycleResponse, InferenceDeleteResponse, InferenceGPUOptionMarket, InferenceGPUOption, InferenceGPUAlternative, InferenceGPUOptionsParams, InferenceGPUOptionsResponse, InferenceCanonicalRequest, InferencePreflightWarning, InferencePreflightResponse, AdapterCompatibility, AdapterCompatibilityResponse, AdapterCompatibilityParams, SupportedArchitecture, ArchitectureScope, SupportedArchitecturesResponse, ApiKey, ApiKeyScope, ApiKeyIntrospection, Organization, OrgMember, OrgInvite, Workspace, Integration, IntegrationCreateParams, StorageUsage, StorageObject, PromptTokensDetails, ChatCompletionUsage, TrainingCadence, TrainingCombinator, ServingKind, TrainingRulePausedReason, LoopTrainingMethod, TrainType, GradersScope, TrainingTrigger, TrainingRunState, TrainingVerdict, TrainingDecision, NotifyState, EvaluationStatus, EvaluationItemStatus, EvaluationWinner, EvaluationWarningCode, ConsentVia, TrainingToolCall, TrainingMessage, TrainingRuleServing, TrainingGPURung, TrainingRule, TrainingRuleConsent, TrainingRulePreflightRefusal, TrainingRuleKeyRef, TrainingRuleKeyCheck, TrainingRulePreflight, TrainingRuleBuildSpec, TrainingRuleTriggerInput, TrainingRuleTrainingInput, TrainingRuleDeployInput, TrainingRuleMoneyInput, TrainingRuleEvaluationInput, TrainingRulePromotionInput, TrainingRuleAcceptTerms, TrainingRulePreflightRequest, TrainingRuleCreateRequest, TrainingRuleUpdateRequest, TrainingRuleConsentRequest, TrainingRuleListParams, TrainingRunPromoteRequest, TrainingRunRejectRequest, TrainingRunRollbackRequest, TrainingRunCancelRequest, TrainingRunListParams, TrainingRun, TrainingRunSummary, TrainingRunEvent, TrainingRunLinks, TrainingRunActions, EvaluationRef, EvaluationDecoding, EvaluationJudgeDimension, EvaluationDimensionScore, EvaluationGraderScore, EvaluationWarning, EvaluationMargin, JudgeAgreementDimension, JudgeAgreement, Evaluation, EvaluationItemSide, EvaluationItem, EvaluationItemListParams, JudgeAgreementParams, AgentSettings, AgentSettingsRequest, BenchmarkStatus, BenchmarkSourceKind, BenchmarkRunStatus, BenchmarkScoring, BenchmarkDimensionScore, Benchmark, BenchmarkItem, BenchmarkRun, BenchmarkHistoryPoint, BenchmarkSource, BenchmarkCreateParams, BenchmarkListParams, BenchmarkItemListParams, BenchmarkHistoryParams, BenchmarkRetireParams, TrainingRuleBenchmarkRequest, InferenceAlias, InferenceAliasRequest, TrainingRuleListResponse, TrainingRuleResponse, TrainingRuleMutationResponse, TrainingRuleDeleteResponse, TrainingRunListResponse, TrainingRunResponse, TrainingRunActionResponse, EvaluationItemsResponse, AgentSettingsResponse, InferenceAliasListResponse, InferenceAliasResponse, InferenceAliasDeleteResponse, BenchmarkListResponse, BenchmarkItemsResponse, BenchmarkHistoryResponse, } from './types.js';
79
79
  /** The closed set of run states a training run never leaves. */
80
80
  export { TERMINAL_RUN_STATES } from './types.js';
81
- /** The most versions a pipeline may be set to make. */
82
- export { PIPELINE_MAX_VERSIONS_CEILING } from './types.js';
package/dist/index.js CHANGED
@@ -36,7 +36,7 @@ import { Loop } from './resources/loop.js';
36
36
  * SDK version. Sent as part of the User-Agent header.
37
37
  * Must match package.json "version" -- enforced by a contract test.
38
38
  */
39
- export const VERSION = '0.2.13-dev.243';
39
+ export const VERSION = '0.2.13-rc.239';
40
40
  export class RunBiOS {
41
41
  /** Search models, fetch configs, check adapter compatibility. */
42
42
  models;
@@ -105,5 +105,3 @@ export { GPU } from './resources/gpu.js';
105
105
  export { Inference, validateChatRequest, parseSSE, buildInferenceRequest, contextSizingBasis, CONTEXT_DEFAULT_CEILING, CONTEXT_EDITABLE_FLOOR, } from './resources/inference.js';
106
106
  /** The closed set of run states a training run never leaves. */
107
107
  export { TERMINAL_RUN_STATES } from './types.js';
108
- /** The most versions a pipeline may be set to make. */
109
- export { PIPELINE_MAX_VERSIONS_CEILING } from './types.js';
@@ -326,7 +326,7 @@ export class Inference {
326
326
  this.key = config.inferenceKey || envInferenceKey() || envApiKey();
327
327
  // Same default host as HttpClient (api.runbios.ai); api.runbios.ai cutover
328
328
  // is planned once its DNS exists — update both call sites together.
329
- this.baseUrl = (config.baseUrl || envBaseUrl() || 'https://api-dev.runbios.ai').replace(/\/+$/, '');
329
+ this.baseUrl = (config.baseUrl || envBaseUrl() || 'https://api-staging.runbios.ai').replace(/\/+$/, '');
330
330
  this.timeout = config.timeout ?? 900_000;
331
331
  this._http = http;
332
332
  }
@@ -1,5 +1,5 @@
1
1
  import type { HttpClient } from '../client.js';
2
- import type { LoopTrace, LoopTraceListResponse, LoopTraceListParams, LoopCaptureParams, LoopCaptureResult, LoopImportParams, LoopImportResult, LoopSignal, LoopSignalParams, LoopCandidate, LoopCandidateParams, LoopGrader, LoopGraderParams, LoopLabel, LoopLabelCount, LoopGradeResult, LoopDataset, LoopDatasetListParams, LoopDatasetItem, LoopDatasetCreateParams, LoopConfig, LoopConfigParams, LoopBuildRule, LoopBuildRuleParams, LoopBuildRuleUpdateParams, LoopStats, LoopJudge, LoopJudgeParams, LoopAgentCredential, LoopAgentStatus, LoopSampleRun, LoopSampleRunParams, LoopJudgeRun, LoopJudgeRunItems, LoopJudgeRunItemStatus, LoopJudgeWork, LoopJudgeVerdict, LoopJudgeVerdictResult, TrainingRule, TrainingRulePreflight, TrainingRulePreflightRequest, TrainingRuleCreateRequest, TrainingRuleUpdateRequest, TrainingRuleConsentRequest, TrainingRuleListParams, TrainingRuleListResponse, TrainingRuleResponse, TrainingRuleMutationResponse, TrainingRuleDeleteResponse, TrainingRun, TrainingRunListParams, TrainingRunListResponse, TrainingRunResponse, TrainingRunActionResponse, TrainingRunPromoteRequest, TrainingRunRejectRequest, TrainingRunRollbackRequest, TrainingRunCancelRequest, Evaluation, EvaluationItemListParams, EvaluationItemsResponse, JudgeAgreement, JudgeAgreementParams, Benchmark, BenchmarkCreateParams, BenchmarkHistoryParams, BenchmarkHistoryResponse, BenchmarkItemListParams, BenchmarkItemsResponse, BenchmarkListParams, BenchmarkListResponse, BenchmarkRetireParams, BenchmarkRun, TrainingRuleBenchmarkRequest, AgentSettings, AgentSettingsRequest, Pipeline, PipelineListResponse } from '../types.js';
2
+ import type { LoopTrace, LoopTraceListResponse, LoopTraceListParams, LoopCaptureParams, LoopCaptureResult, LoopImportParams, LoopImportResult, LoopSignal, LoopSignalParams, LoopCandidate, LoopCandidateParams, LoopGrader, LoopGraderParams, LoopLabel, LoopLabelCount, LoopGradeResult, LoopDataset, LoopDatasetListParams, LoopDatasetItem, LoopDatasetCreateParams, LoopConfig, LoopConfigParams, LoopBuildRule, LoopBuildRuleParams, LoopBuildRuleUpdateParams, LoopStats, LoopJudge, LoopJudgeParams, LoopAgentCredential, LoopAgentStatus, LoopSampleRun, LoopSampleRunParams, LoopJudgeRun, LoopJudgeRunItems, LoopJudgeRunItemStatus, LoopJudgeWork, LoopJudgeVerdict, LoopJudgeVerdictResult, TrainingRule, TrainingRulePreflight, TrainingRulePreflightRequest, TrainingRuleCreateRequest, TrainingRuleUpdateRequest, TrainingRuleConsentRequest, TrainingRuleListParams, TrainingRuleListResponse, TrainingRuleResponse, TrainingRuleMutationResponse, TrainingRuleDeleteResponse, TrainingRun, TrainingRunListParams, TrainingRunListResponse, TrainingRunResponse, TrainingRunActionResponse, TrainingRunPromoteRequest, TrainingRunRejectRequest, TrainingRunRollbackRequest, TrainingRunCancelRequest, Evaluation, EvaluationItemListParams, EvaluationItemsResponse, JudgeAgreement, JudgeAgreementParams, Benchmark, BenchmarkCreateParams, BenchmarkHistoryParams, BenchmarkHistoryResponse, BenchmarkItemListParams, BenchmarkItemsResponse, BenchmarkListParams, BenchmarkListResponse, BenchmarkRetireParams, BenchmarkRun, TrainingRuleBenchmarkRequest, AgentSettings, AgentSettingsRequest } from '../types.js';
3
3
  /**
4
4
  * The Conscious Loop -- capture what your model was asked and answered, record
5
5
  * whether it was right, and turn those judgements into training data.
@@ -335,10 +335,7 @@ export declare class Loop {
335
335
  * Delete a training set.
336
336
  *
337
337
  * The conversations it was built from are untouched -- a set is a selection,
338
- * and discarding the selection must not discard the evidence. A set a
339
- * training run was trained on is refused `409 DATASET_IN_USE` and kept, as
340
- * the record of what that run learned from. See
341
- * `LoopDatasetDeleteRefusalCode`.
338
+ * and discarding the selection must not discard the evidence.
342
339
  */
343
340
  deleteDataset(id: string): Promise<{
344
341
  deleted: boolean;
@@ -617,12 +614,11 @@ export declare class Loop {
617
614
  */
618
615
  listTrainingRules(params?: TrainingRuleListParams): Promise<TrainingRuleListResponse>;
619
616
  /**
620
- * Read one rule with its recent runs, what its runs this month have cost at
621
- * most, and how far its judge agrees with your own reviewers.
617
+ * Read one rule with its recent runs, what it has spent this month, and how
618
+ * far its judge agrees with your own reviewers.
622
619
  *
623
620
  * Returned whole rather than unwrapped to the rule: `month_spent_cents` is
624
- * the number that says whether the monthly ceiling is about to stop it. It
625
- * is an upper bound, not an exact spend: see `Pipeline.month_spent_cents`.
621
+ * the number that says whether the monthly ceiling is about to stop it.
626
622
  */
627
623
  getTrainingRule(id: string): Promise<TrainingRuleResponse>;
628
624
  /**
@@ -630,11 +626,8 @@ export declare class Loop {
630
626
  * that can be cleared.
631
627
  *
632
628
  * A money-bearing change -- a ceiling, a model, a GPU ladder, the promotion
633
- * policy, or changing `explore_recipes` in EITHER direction -- bumps
634
- * `revision`, clears the recorded consent and STOPS the rule firing until
635
- * someone accepts the new amounts. Turning recipe variants OFF does this
636
- * too: the pipeline then makes no version at all, variant or not, until the
637
- * terms are accepted again. The reply says so in
629
+ * policy -- bumps `revision`, clears the recorded consent and STOPS the rule
630
+ * firing until someone accepts the new amounts. The reply says so in
638
631
  * `consent_required`, and carries a fresh `preflight` with the new figures.
639
632
  * Pass `expected_revision` to be refused with `REVISION_MISMATCH` rather
640
633
  * than overwrite an edit somebody else made in the meantime.
@@ -661,17 +654,8 @@ export declare class Loop {
661
654
  * Fire a rule now, without waiting for its cadence.
662
655
  *
663
656
  * Bypasses the schedule and `min_new_rows` only. The row floors, the money
664
- * ceilings, the consent, the version limit and the monthly limit all still
665
- * apply, so this can answer `409 CONSENT_REQUIRED`, `409 RULE_PAUSED`,
666
- * `409 RUN_ACTIVE`, `409 VERSION_LIMIT_REACHED` (the pipeline has made
667
- * `max_versions` versions), `409 MONTHLY_LIMIT_REACHED` (the most this
668
- * month's runs can have cost plus the most one run may cost would pass the
669
- * monthly limit; the error
670
- * body carries `month_spent_cents`, `monthly_ceiling_cents`, `run_max_cents`
671
- * and `resumes_at`), `409 AGENT_OFF` (the workspace's Conscious Loop agent
672
- * is not turned on, so a trained model could not be compared) or
673
- * `422 NOT_ENOUGH_ROWS` with the counts it needed. See
674
- * `TrainingRuleRunRefusalCode`.
657
+ * ceilings and the consent all still apply, so this can answer `409
658
+ * CONSENT_REQUIRED` or `422 NOT_ENOUGH_ROWS` with the counts it needed.
675
659
  */
676
660
  runTrainingRule(id: string): Promise<TrainingRun>;
677
661
  /**
@@ -726,34 +710,6 @@ export declare class Loop {
726
710
  * cancelling a training job does not refund the hours it burned.
727
711
  */
728
712
  cancelTrainingRun(id: string, params?: TrainingRunCancelRequest): Promise<TrainingRunActionResponse>;
729
- /**
730
- * Every training rule in the workspace, seen as the series of versions it
731
- * produced: which exist, which one serves (`champion_version`), how each did
732
- * against the champion of its day and on the standing benchmark, the run in
733
- * flight and which version it will be, and what the pipeline is waiting for.
734
- *
735
- * Read-only. Every decision stays on the route that owns it -- promote,
736
- * reject and roll back on the run, the version limit and `explore_recipes`
737
- * on the rule.
738
- *
739
- * RETURNED WITH ITS ENVELOPE, because the list is not always all of them:
740
- * the service returns the newest 100, `total` is every pipeline in the
741
- * workspace and `truncated` is true when more exist than were returned. A
742
- * caller that shows `pipelines` alone presents a short list as the whole of
743
- * it. The rest are reached through `listTrainingRules`, which pages.
744
- */
745
- listPipelines(): Promise<PipelineListResponse>;
746
- /**
747
- * One pipeline, by its training rule's id.
748
- *
749
- * `status` says whether it is the platform working (`running`), the member
750
- * who has to act (`needs_review`, `needs_funds`), or nothing at all until
751
- * somebody does (`paused`, with `paused_reason` or `needs_consent` saying
752
- * why; `complete` at `max_versions`). `month_spent_cents` is the figure the
753
- * monthly limit is enforced against: what runs started this month have cost
754
- * at most, not an exact spend.
755
- */
756
- getPipeline(id: string): Promise<Pipeline>;
757
713
  /**
758
714
  * The comparison behind a verdict: both models on the same held-out rows,
759
715
  * with identical decoding, judge and grader names resolved.
@@ -440,10 +440,7 @@ export class Loop {
440
440
  * Delete a training set.
441
441
  *
442
442
  * The conversations it was built from are untouched -- a set is a selection,
443
- * and discarding the selection must not discard the evidence. A set a
444
- * training run was trained on is refused `409 DATASET_IN_USE` and kept, as
445
- * the record of what that run learned from. See
446
- * `LoopDatasetDeleteRefusalCode`.
443
+ * and discarding the selection must not discard the evidence.
447
444
  */
448
445
  async deleteDataset(id) {
449
446
  return this._http.fetchDelete(`/api/loop/datasets/${encodeURIComponent(id)}`);
@@ -849,12 +846,11 @@ export class Loop {
849
846
  return this._http.fetchGet(`/api/loop/training-rules${qs ? `?${qs}` : ''}`);
850
847
  }
851
848
  /**
852
- * Read one rule with its recent runs, what its runs this month have cost at
853
- * most, and how far its judge agrees with your own reviewers.
849
+ * Read one rule with its recent runs, what it has spent this month, and how
850
+ * far its judge agrees with your own reviewers.
854
851
  *
855
852
  * Returned whole rather than unwrapped to the rule: `month_spent_cents` is
856
- * the number that says whether the monthly ceiling is about to stop it. It
857
- * is an upper bound, not an exact spend: see `Pipeline.month_spent_cents`.
853
+ * the number that says whether the monthly ceiling is about to stop it.
858
854
  */
859
855
  async getTrainingRule(id) {
860
856
  return this._http.fetchGet(`/api/loop/training-rules/${encodeURIComponent(id)}`);
@@ -864,11 +860,8 @@ export class Loop {
864
860
  * that can be cleared.
865
861
  *
866
862
  * A money-bearing change -- a ceiling, a model, a GPU ladder, the promotion
867
- * policy, or changing `explore_recipes` in EITHER direction -- bumps
868
- * `revision`, clears the recorded consent and STOPS the rule firing until
869
- * someone accepts the new amounts. Turning recipe variants OFF does this
870
- * too: the pipeline then makes no version at all, variant or not, until the
871
- * terms are accepted again. The reply says so in
863
+ * policy -- bumps `revision`, clears the recorded consent and STOPS the rule
864
+ * firing until someone accepts the new amounts. The reply says so in
872
865
  * `consent_required`, and carries a fresh `preflight` with the new figures.
873
866
  * Pass `expected_revision` to be refused with `REVISION_MISMATCH` rather
874
867
  * than overwrite an edit somebody else made in the meantime.
@@ -900,17 +893,8 @@ export class Loop {
900
893
  * Fire a rule now, without waiting for its cadence.
901
894
  *
902
895
  * Bypasses the schedule and `min_new_rows` only. The row floors, the money
903
- * ceilings, the consent, the version limit and the monthly limit all still
904
- * apply, so this can answer `409 CONSENT_REQUIRED`, `409 RULE_PAUSED`,
905
- * `409 RUN_ACTIVE`, `409 VERSION_LIMIT_REACHED` (the pipeline has made
906
- * `max_versions` versions), `409 MONTHLY_LIMIT_REACHED` (the most this
907
- * month's runs can have cost plus the most one run may cost would pass the
908
- * monthly limit; the error
909
- * body carries `month_spent_cents`, `monthly_ceiling_cents`, `run_max_cents`
910
- * and `resumes_at`), `409 AGENT_OFF` (the workspace's Conscious Loop agent
911
- * is not turned on, so a trained model could not be compared) or
912
- * `422 NOT_ENOUGH_ROWS` with the counts it needed. See
913
- * `TrainingRuleRunRefusalCode`.
896
+ * ceilings and the consent all still apply, so this can answer `409
897
+ * CONSENT_REQUIRED` or `422 NOT_ENOUGH_ROWS` with the counts it needed.
914
898
  */
915
899
  async runTrainingRule(id) {
916
900
  const res = await this._http.fetchPost(`/api/loop/training-rules/${encodeURIComponent(id)}/run`, {});
@@ -996,47 +980,6 @@ export class Loop {
996
980
  async cancelTrainingRun(id, params = {}) {
997
981
  return this._http.fetchPost(`/api/loop/training-runs/${encodeURIComponent(id)}/cancel`, params);
998
982
  }
999
- // ── pipelines ─────────────────────────────────────────────────────────
1000
- /**
1001
- * Every training rule in the workspace, seen as the series of versions it
1002
- * produced: which exist, which one serves (`champion_version`), how each did
1003
- * against the champion of its day and on the standing benchmark, the run in
1004
- * flight and which version it will be, and what the pipeline is waiting for.
1005
- *
1006
- * Read-only. Every decision stays on the route that owns it -- promote,
1007
- * reject and roll back on the run, the version limit and `explore_recipes`
1008
- * on the rule.
1009
- *
1010
- * RETURNED WITH ITS ENVELOPE, because the list is not always all of them:
1011
- * the service returns the newest 100, `total` is every pipeline in the
1012
- * workspace and `truncated` is true when more exist than were returned. A
1013
- * caller that shows `pipelines` alone presents a short list as the whole of
1014
- * it. The rest are reached through `listTrainingRules`, which pages.
1015
- */
1016
- async listPipelines() {
1017
- const res = await this._http.fetchGet('/api/loop/pipelines');
1018
- const pipelines = res.pipelines || [];
1019
- return {
1020
- pipelines,
1021
- // An older service sends neither: its list is taken at its word.
1022
- total: typeof res.total === 'number' ? res.total : pipelines.length,
1023
- truncated: res.truncated === true,
1024
- };
1025
- }
1026
- /**
1027
- * One pipeline, by its training rule's id.
1028
- *
1029
- * `status` says whether it is the platform working (`running`), the member
1030
- * who has to act (`needs_review`, `needs_funds`), or nothing at all until
1031
- * somebody does (`paused`, with `paused_reason` or `needs_consent` saying
1032
- * why; `complete` at `max_versions`). `month_spent_cents` is the figure the
1033
- * monthly limit is enforced against: what runs started this month have cost
1034
- * at most, not an exact spend.
1035
- */
1036
- async getPipeline(id) {
1037
- const res = await this._http.fetchGet(`/api/loop/pipelines/${encodeURIComponent(id)}`);
1038
- return res.pipeline;
1039
- }
1040
983
  // ── the comparison report ─────────────────────────────────────────────
1041
984
  /**
1042
985
  * The comparison behind a verdict: both models on the same held-out rows,
package/dist/types.d.ts CHANGED
@@ -8,13 +8,13 @@ export interface BiOSConfig {
8
8
  orgId?: string;
9
9
  /** Workspace ID. Optional when using API keys (resolved from the key). Can override for multi-workspace keys. */
10
10
  workspaceId?: string;
11
- /** Base URL for the API. Falls back to the RUNBIOS_BASE_URL environment variable (legacy: BIOS_BASE_URL), then the canonical https://api-dev.runbios.ai hostname. */
11
+ /** Base URL for the API. Falls back to the RUNBIOS_BASE_URL environment variable (legacy: BIOS_BASE_URL), then the canonical https://api-staging.runbios.ai hostname. */
12
12
  baseUrl?: string;
13
13
  /** Request timeout in milliseconds. Defaults to 30000. */
14
14
  timeout?: number;
15
15
  /** Default per-deployment inference key. Can be overridden per inference call. */
16
16
  inferenceKey?: string;
17
- /** Inference base URL. Defaults to baseUrl, then https://api-dev.runbios.ai. */
17
+ /** Inference base URL. Defaults to baseUrl, then https://api-staging.runbios.ai. */
18
18
  inferenceBaseUrl?: string;
19
19
  /** End-to-end inference timeout in milliseconds. Defaults to 15 minutes. */
20
20
  inferenceTimeout?: number;
@@ -2533,28 +2533,6 @@ export interface LoopSignalParams {
2533
2533
  reason?: string;
2534
2534
  author?: string;
2535
2535
  metadata?: Record<string, unknown>;
2536
- /**
2537
- * Which training pipeline this feedback feeds, named in the SAME call.
2538
- *
2539
- * A build rule selects conversations by label and a training rule trains from
2540
- * that build rule, so a label IS the pipeline a conversation goes down.
2541
- * Putting one on used to need a second request after this one — and that
2542
- * second request is the one that gets skipped, by a script that handles the
2543
- * 201 and moves on, by a retry that succeeds here and fails there, by an
2544
- * integration whose author never knew labelling was a step.
2545
- *
2546
- * What it leaves behind is a reviewed conversation in no pipeline: counted in
2547
- * every "reviewed" total and selected by nothing.
2548
- *
2549
- * Written in one transaction with the verdict: either both land or neither
2550
- * does. Omit them and nothing changes — the overwhelming majority of feedback
2551
- * names no pipeline and behaves exactly as it always has.
2552
- */
2553
- labels?: string[];
2554
- /** The same, by dimension: `{ category: 'billing' }`. */
2555
- attributes?: Record<string, string>;
2556
- /** The master group the bare labels belong to. */
2557
- parent?: string;
2558
2536
  }
2559
2537
  export interface LoopTrace {
2560
2538
  id: string;
@@ -3147,46 +3125,9 @@ export type TrainingRulePausedReason = 'consent_invalid' | 'insufficient_funds'
3147
3125
  * republished to keep up with a server-side capability flag.
3148
3126
  */
3149
3127
  export type LoopTrainingMethod = 'sft' | 'rlhf';
3150
- /**
3151
- * What a rule's stored `train_type` may READ as.
3152
- *
3153
- * Still three, and deliberately: the column has always allowed `full`, so a
3154
- * rule created before standing rules were narrowed to adapters can still come
3155
- * back carrying it. Narrowing the response type would make this SDK
3156
- * misrepresent a row that really exists.
3157
- */
3158
3128
  export type TrainType = 'lora' | 'qlora' | 'full';
3159
- /**
3160
- * What a rule may be SET to, which is narrower.
3161
- *
3162
- * loop-service refuses `full` on preflight, create and update for a standing
3163
- * rule: a rule trains unattended and on repeat, and a full fine-tune rewrites
3164
- * every weight instead of adding a small adapter, so it cannot be compared or
3165
- * rolled back cheaply. Typing the request as the wider set handed callers a
3166
- * value guaranteed to come back a 400. One-off full fine-tunes are unaffected
3167
- * — they are a different API.
3168
- */
3169
- export type RuleTrainType = 'lora' | 'qlora';
3170
3129
  export type GradersScope = 'source' | 'all' | 'none';
3171
- /**
3172
- * Why a run fired. `variant` is a recipe variant: the pipeline had nothing new
3173
- * to learn, so it trained the same base model on the same conversations with
3174
- * one training setting changed (see {@link TrainingRecipe}).
3175
- */
3176
- export type TrainingTrigger = 'rows' | 'cadence' | 'both' | 'manual' | 'variant';
3177
- /**
3178
- * The four training settings a recipe variant may change, as the run was
3179
- * actually trained: the training service's own defaults are filled in where
3180
- * the rule left a knob unset, so two recipes can be compared value for value.
3181
- * A knob that does not apply to the run (a LoRA rank on a full fine-tune) is
3182
- * null, never a guessed number.
3183
- */
3184
- export interface TrainingRecipe {
3185
- learning_rate: number | null;
3186
- num_train_epochs: number | null;
3187
- lora_rank: number | null;
3188
- lora_alpha: number | null;
3189
- }
3130
+ export type TrainingTrigger = 'rows' | 'cadence' | 'both' | 'manual';
3190
3131
  export type TrainingRunState = 'built' | 'registering' | 'dataset_validating' | 'preflighting' | 'job_creating' | 'training' | 'checkpoint_ready' | 'candidate_booking' | 'candidate_running' | 'evaluating' | 'reported' | 'awaiting_review' | 'promoting' | 'waiting_funds' | 'promoted' | 'rejected' | 'expired' | 'failed' | 'budget_stopped' | 'cancelled' | 'superseded' | 'rolled_back';
3191
3132
  /** The closed set a run never leaves. */
3192
3133
  export declare const TERMINAL_RUN_STATES: readonly TrainingRunState[];
@@ -3277,30 +3218,6 @@ export interface TrainingRule {
3277
3218
  eval_max_rows: number;
3278
3219
  eval_max_tokens: number;
3279
3220
  min_holdout_rows: number;
3280
- /**
3281
- * How many versions this pipeline may make, 1 to 10 (five unless changed).
3282
- * A version is a trained model whose comparison was reported having scored
3283
- * at least one conversation, or one a member put live; a run that failed,
3284
- * was cancelled or compared nothing is not one and does not count. At the
3285
- * limit the pipeline stops, and a manual run is refused with
3286
- * `409 VERSION_LIMIT_REACHED`, until the limit is raised.
3287
- */
3288
- max_versions: number;
3289
- /**
3290
- * Try other training recipes when there is nothing new to learn.
3291
- *
3292
- * MONEY-BEARING. When on, a pipeline that has made at least one version, and
3293
- * has nothing reviewed since it that adds anything new to train on (nothing
3294
- * reviewed at all, or only reviews -- thumbs-down with no correction, say --
3295
- * that give it no row the last version's set lacked), trains the same base
3296
- * model on the same conversations with ONE setting changed, and that model
3297
- * takes over only if it beats what serves the app, like any other version.
3298
- * Each try is a full paid run inside the rule's per-run ceilings and its
3299
- * monthly limit, so changing this in EITHER direction changes the terms: the
3300
- * rule stops firing until the member accepts them again. False unless
3301
- * somebody turned it on.
3302
- */
3303
- explore_recipes: boolean;
3304
3221
  /**
3305
3222
  * The standing benchmark replayed on every run of this rule, beside the
3306
3223
  * per-run comparison and never instead of it. Null is the ordinary state.
@@ -3310,32 +3227,6 @@ export interface TrainingRule {
3310
3227
  * its own. It raises no amount, so it does not invalidate consent.
3311
3228
  */
3312
3229
  benchmark_id: string | null;
3313
- /**
3314
- * Whether that benchmark DECIDES the verdict, or only reports a number.
3315
- *
3316
- * False for every rule that has not asked, which is the point of it being
3317
- * separate from `benchmark_id`. Attaching a benchmark means "replay this
3318
- * fixed set on every run and put the score on the report"; it does not mean
3319
- * "let that score overrule the comparison this rule was built on".
3320
- *
3321
- * When it is on it cuts both ways. A run whose held-out split came up short
3322
- * can be decided at all -- before this, such a run trained a model, billed a
3323
- * GPU and could never promote -- and a candidate that wins on fresh
3324
- * conversations while losing ground on the fixed set is refused.
3325
- */
3326
- benchmark_decides: boolean;
3327
- /**
3328
- * How many of the benchmark's pinned conversations must have scored before
3329
- * it may decide anything. A replay that got through four of its forty has
3330
- * not measured the new model.
3331
- */
3332
- benchmark_min_rows: number;
3333
- /**
3334
- * The bar on the replay's own candidate-minus-incumbent delta. Like-for-like
3335
- * WITHIN one replay -- both models, same pinned rows, same frozen judge, one
3336
- * pass -- and not comparable between runs.
3337
- */
3338
- benchmark_min_delta: number;
3339
3230
  auto_promote: boolean;
3340
3231
  promote_margin: number;
3341
3232
  promote_min_win_rate: number;
@@ -3416,14 +3307,6 @@ export interface TrainingRulePreflight {
3416
3307
  key_check: TrainingRuleKeyCheck;
3417
3308
  terms_text: string;
3418
3309
  terms_version: string;
3419
- /**
3420
- * The machines the hourly amounts are per hour of: the ladders the platform
3421
- * chose when the request left them empty, or the caller's own echoed back
3422
- * unchanged. An amount per hour means nothing without knowing what it is per
3423
- * hour of, so show these beside the estimate before anyone accepts it.
3424
- */
3425
- train_gpu_priorities: TrainingGPURung[];
3426
- deploy_gpu_priorities: TrainingGPURung[];
3427
3310
  }
3428
3311
  export interface TrainingRuleBuildSpec {
3429
3312
  method: string;
@@ -3457,18 +3340,6 @@ export interface TrainingRuleTriggerInput {
3457
3340
  combinator?: TrainingCombinator;
3458
3341
  /** null removes the row floor, leaving the schedule as the only trigger. */
3459
3342
  min_new_rows?: number | null;
3460
- /** 1 to 10. Absent leaves it as it is; absent on a create means five. */
3461
- max_versions?: number;
3462
- /**
3463
- * See {@link TrainingRule.explore_recipes}: every variant it allows is a
3464
- * full paid run, so changing it -- on OR off -- is a money-bearing edit that
3465
- * pauses the rule, and it makes no version of any kind, until the new terms
3466
- * are accepted. Turning it off to save money stops the pipeline too, until
3467
- * somebody accepts again. It sits beside `max_versions`
3468
- * because both say when the pipeline makes another version. Absent leaves
3469
- * it as it is; absent on a create means off.
3470
- */
3471
- explore_recipes?: boolean;
3472
3343
  }
3473
3344
  export interface TrainingRuleTrainingInput {
3474
3345
  model_id?: string;
@@ -3480,7 +3351,7 @@ export interface TrainingRuleTrainingInput {
3480
3351
  * plain supervised training means.
3481
3352
  */
3482
3353
  rlhf_type?: string | null;
3483
- train_type?: RuleTrainType;
3354
+ train_type?: TrainType;
3484
3355
  config?: Record<string, unknown>;
3485
3356
  train_gpu_priorities?: TrainingGPURung[];
3486
3357
  train_max_price_hour_cents?: number;
@@ -3563,19 +3434,6 @@ export interface TrainingRuleCreateRequest extends TrainingRulePreflightRequest
3563
3434
  * preflight body and not on the update body.
3564
3435
  */
3565
3436
  benchmark_id?: string;
3566
- /**
3567
- * And whether that benchmark decides, for the same reason `benchmark_id` is
3568
- * here: run 1 reads its gate off the snapshot it freezes, and a gate applied
3569
- * by a second call lands after run 1 has been decided without it.
3570
- *
3571
- * `400 BENCHMARK_GATE_HAS_NO_SET` when it is true with no benchmark,
3572
- * `400 BENCHMARK_GATE_NEEDS_MIN_ROWS` when it is true with no floor.
3573
- */
3574
- benchmark_decides?: boolean;
3575
- /** Cannot exceed the set's size: `400 BENCHMARK_GATE_UNREACHABLE`. */
3576
- benchmark_min_rows?: number;
3577
- /** Zero -- the default -- means "must not lose ground". */
3578
- benchmark_min_delta?: number;
3579
3437
  }
3580
3438
  export interface TrainingRuleUpdateRequest extends TrainingRulePreflightRequest {
3581
3439
  expected_revision?: number;
@@ -3621,18 +3479,6 @@ export interface TrainingRun {
3621
3479
  authorized_by: string;
3622
3480
  rule_snapshot: Record<string, unknown>;
3623
3481
  trigger: TrainingTrigger;
3624
- /**
3625
- * For a recipe variant, what it changed, in plain words ("half the learning
3626
- * rate"). Null for every run that trained on the ordinary recipe.
3627
- */
3628
- recipe_note: string | null;
3629
- /**
3630
- * The version whose recipe this run varied. Null when the variant started
3631
- * from the rule's own settings, and for every run that is not a variant.
3632
- */
3633
- recipe_of_version: number | null;
3634
- /** The four recipe knobs this run trained with, defaults filled in. */
3635
- recipe: TrainingRecipe;
3636
3482
  fired_reason: string | null;
3637
3483
  state: TrainingRunState;
3638
3484
  state_entered_at: string;
@@ -3641,13 +3487,6 @@ export interface TrainingRun {
3641
3487
  last_reason: string | null;
3642
3488
  error_code: string | null;
3643
3489
  last_error: string | null;
3644
- /**
3645
- * What to DO about the failure, in one sentence, derived server-side from
3646
- * `error_code` -- the same words the failure email carries, so the two
3647
- * cannot drift. Null when the run has not failed, or when the failure has
3648
- * no remedy worth printing.
3649
- */
3650
- error_next_step: string | null;
3651
3490
  training_ceiling_cents: number;
3652
3491
  candidate_ceiling_cents: number;
3653
3492
  eval_ceiling_cents: number;
@@ -3664,13 +3503,6 @@ export interface TrainingRun {
3664
3503
  queue_deadline_at: string | null;
3665
3504
  checkpoint_id: string | null;
3666
3505
  checkpoint_step: number | null;
3667
- /**
3668
- * Which version of the pipeline this run produced. Null for a run that is
3669
- * not a version: one that failed, was cancelled, or whose comparison scored
3670
- * nothing, and that nobody put live. `seq` counts attempts; this counts
3671
- * models that competed, plus any a member promoted without a comparison.
3672
- */
3673
- version_no: number | null;
3674
3506
  training_eval_loss: number | null;
3675
3507
  candidate_deployment_id: string | null;
3676
3508
  candidate_name: string | null;
@@ -3714,20 +3546,10 @@ export interface TrainingRunSummary {
3714
3546
  workspace_id: string;
3715
3547
  seq: number;
3716
3548
  trigger: TrainingTrigger;
3717
- /** See TrainingRun.recipe_note. On the list row too: the Runs table is where a variant is first seen. */
3718
- recipe_note: string | null;
3719
- /** See TrainingRun.recipe_of_version. */
3720
- recipe_of_version: number | null;
3721
- /** See TrainingRun.recipe. */
3722
- recipe: TrainingRecipe;
3723
3549
  state: TrainingRunState;
3724
3550
  state_entered_at: string;
3725
3551
  last_reason: string | null;
3726
3552
  error_code: string | null;
3727
- /** See TrainingRun.error_next_step. On the list row too: the Runs table is where a failure is first seen. */
3728
- error_next_step: string | null;
3729
- /** See TrainingRun.version_no. */
3730
- version_no: number | null;
3731
3553
  verdict: TrainingVerdict | null;
3732
3554
  decision: TrainingDecision | null;
3733
3555
  train_rows: number | null;
@@ -4153,290 +3975,13 @@ export interface BenchmarkRetireParams {
4153
3975
  reason?: string;
4154
3976
  }
4155
3977
  /**
4156
- * Attach a benchmark to a rule, detach it with a present null, or move the bar
4157
- * it decides on.
4158
- *
4159
- * ABSENT IS "LEAVE IT", PRESENT IS "MAKE IT THIS", and that applies to every
4160
- * field here. This is the route that attaches a benchmark AND the route that
4161
- * moves its bar a month later, so a call naming only `benchmark_id` must not
4162
- * reset a floor somebody chose, and a call naming only `benchmark_min_delta`
4163
- * must not detach the set.
3978
+ * Attach a benchmark to a rule, or detach it with a present null.
4164
3979
  *
4165
- * Detaching is the one exception: a present null `benchmark_id` clears
4166
- * `benchmark_decides` with it, because a gate with nothing behind it is a
4167
- * verdict waiting on a measurement that will never come. Asking for both in
4168
- * one request -- null id and `benchmark_decides: true` -- is refused rather
4169
- * than resolved, because either resolution is a guess about which half you
4170
- * meant.
3980
+ * Not optional, and not omittable: this route sets the field, so an absent key
3981
+ * would be a request with nothing in it. Send the id to attach, null to detach.
4171
3982
  */
4172
3983
  export interface TrainingRuleBenchmarkRequest {
4173
- benchmark_id?: string | null;
4174
- benchmark_decides?: boolean;
4175
- benchmark_min_rows?: number;
4176
- benchmark_min_delta?: number;
4177
- }
4178
- /**
4179
- * The machine codes `runTrainingRule` can be refused with. Each is the
4180
- * platform declining to start a paid run it would only refuse a minute later,
4181
- * so each is said in front of the caller rather than left for the rule's
4182
- * `last_reason` to explain after they have stopped looking.
4183
- *
4184
- * - `CONSENT_REQUIRED` (409): the terms changed and nobody accepted them.
4185
- * - `RULE_PAUSED` (409): the rule is switched off or the platform stopped it.
4186
- * - `RUN_ACTIVE` (409): one run at a time, so two cannot race to promote.
4187
- * - `VERSION_LIMIT_REACHED` (409): the pipeline has made `max_versions`
4188
- * versions. Raise the limit or start a new pipeline.
4189
- * - `MONTHLY_LIMIT_REACHED` (409): the most this month's runs can have cost
4190
- * plus the most one more run may cost would pass `monthly_ceiling_cents`.
4191
- * The body carries the figures:
4192
- * see {@link TrainingRuleMonthlyLimitRefusal}.
4193
- * - `AGENT_OFF` (409): the workspace's Conscious Loop agent is not turned on.
4194
- * Every comparison runs under the agent's key, so a run started without one
4195
- * would pay for training and could not be compared. Turn it on in Agent
4196
- * settings, then run the rule.
4197
- * - `NOT_ENOUGH_ROWS` (422): too few reviewed or held-out conversations, with
4198
- * `train_rows`, `holdout_rows` and the `floor` it needed.
4199
- */
4200
- export type TrainingRuleRunRefusalCode = 'CONSENT_REQUIRED' | 'RULE_PAUSED' | 'RUN_ACTIVE' | 'VERSION_LIMIT_REACHED' | 'MONTHLY_LIMIT_REACHED' | 'AGENT_OFF' | 'NOT_ENOUGH_ROWS';
4201
- /**
4202
- * The machine codes `deleteDataset` can be refused with.
4203
- *
4204
- * - `DATASET_IN_USE` (409): a training run was trained on this set, so it is
4205
- * kept as the record of what that run learned from.
4206
- */
4207
- export type LoopDatasetDeleteRefusalCode = 'DATASET_IN_USE';
4208
- /**
4209
- * The body of a manual run refused `409 MONTHLY_LIMIT_REACHED`, as
4210
- * `ApiError.body` carries it.
4211
- *
4212
- * The test is the worst case, not the average: a run may start only if the
4213
- * most the rule's runs this UTC calendar month can have cost
4214
- * (`month_spent_cents`), plus the most the run about to start may spend (its
4215
- * three ceilings added up), fits under the limit.
4216
- */
4217
- export interface TrainingRuleMonthlyLimitRefusal {
4218
- error: {
4219
- code: 'MONTHLY_LIMIT_REACHED';
4220
- message: string;
4221
- };
4222
- /** What this rule's runs have cost at most this UTC calendar month; see {@link Pipeline.month_spent_cents}. */
4223
- month_spent_cents: number;
4224
- /** The rule's monthly limit. */
4225
- monthly_ceiling_cents: number;
4226
- /** The most one run of this rule may spend: training + candidate + evaluation ceilings. */
4227
- run_max_cents: number;
4228
- /**
4229
- * The first instant of the next UTC month, when the counted spend starts
4230
- * again from nothing. When `run_max_cents` alone exceeds the limit, no month
4231
- * will ever fit and only raising the limit helps; `message` says which.
4232
- */
4233
- resumes_at: string;
4234
- }
4235
- /** The most versions a pipeline may be set to make (loop-service maxVersionsCeiling). */
4236
- export declare const PIPELINE_MAX_VERSIONS_CEILING = 10;
4237
- /**
4238
- * What a pipeline is doing. `needs_review` and `needs_funds` are a finished
4239
- * version waiting on the MEMBER -- a decision, or a top-up -- which is not the
4240
- * platform working on it.
4241
- *
4242
- * `paused` is a rule that is switched off, AND an enabled rule the platform
4243
- * has stopped firing (`paused_reason`) or whose terms changed and were never
4244
- * accepted (`needs_consent`). `complete` is a pipeline at its `max_versions`.
4245
- */
4246
- export type PipelineStatus = 'running' | 'needs_review' | 'needs_funds' | 'waiting' | 'paused' | 'complete';
4247
- /**
4248
- * The run in flight. It is not a version until its comparison is reported
4249
- * having scored at least one conversation, or a member puts it live.
4250
- */
4251
- export interface PipelineActiveRun {
4252
- run_id: string;
4253
- state: TrainingRunState;
4254
- since: string;
4255
- /**
4256
- * The version number it will take if it becomes one: if its comparison
4257
- * scores something, or if a member promotes it anyway.
4258
- */
4259
- will_be_version: number;
4260
- version_no: number | null;
4261
- last_reason: string | null;
4262
- last_error: string | null;
4263
- train_rows: number | null;
4264
- holdout_rows: number | null;
4265
- /**
4266
- * Its comparison scored at least one conversation. False on a reported run
4267
- * means the held-back set was too small to compare anything: that run is
4268
- * not a version, and becomes one only if a member promotes it.
4269
- */
4270
- competed: boolean;
4271
- /** What recipe it is trying, when it is a recipe variant. See TrainingRun.recipe_note. */
4272
- recipe_note: string | null;
4273
- }
4274
- /**
4275
- * One version: a trained model whose comparison scored at least one
4276
- * conversation, or one a member put live without one. The second kind has
4277
- * `rows_scored` 0 and `win_rate` null, and may be the champion.
4278
- */
4279
- export interface PipelineVersion {
4280
- version: number;
4281
- run_id: string;
4282
- seq: number;
4283
- state: TrainingRunState;
4284
- verdict: TrainingVerdict | null;
4285
- decision: TrainingDecision | null;
4286
- decided_at: string | null;
4287
- /** True for the version the app is served by today. */
4288
- is_champion: boolean;
4289
- checkpoint_id: string | null;
4290
- train_rows: number | null;
4291
- holdout_rows: number | null;
4292
- /** Head-to-head against the champion of its day, on conversations neither trained on. */
4293
- rows_scored: number | null;
4294
- win_rate: number | null;
4295
- mean_delta: number | null;
4296
- /**
4297
- * The standing benchmark's replay of this version, when the rule has one.
4298
- * Only scores with `benchmark_comparable` sit on one axis.
4299
- */
4300
- benchmark_score: number | null;
4301
- benchmark_rows: number | null;
4302
- benchmark_rows_total: number | null;
4303
- benchmark_status: string | null;
4304
- /**
4305
- * False for a score that is not on today's yardstick: measured on an earlier
4306
- * revision of the benchmark's items, or on a different benchmark than the
4307
- * one attached now. Never draw or subtract those on the same line.
4308
- */
4309
- benchmark_comparable: boolean;
4310
- /**
4311
- * Everything this version's run cost, in cents: training, the comparison
4312
- * machine, the judge and the benchmark replay.
4313
- */
4314
- cost_cents: number;
4315
- /**
4316
- * True when no charge was recorded for the comparison machine (a run made
4317
- * before the platform recorded one) and `cost_cents` counts it at its
4318
- * ceiling instead, so the figure is the most the version can have cost,
4319
- * not what it did.
4320
- */
4321
- cost_includes_candidate_ceiling: boolean;
4322
- created_at: string;
4323
- /**
4324
- * When this version's head-to-head started. The champion it faced is the
4325
- * model that was serving at that moment, which is why a delta against "the
4326
- * champion before it" is decided by this and not by `created_at`. Null when
4327
- * no comparison was ever started for it.
4328
- */
4329
- compared_at: string | null;
4330
- /**
4331
- * Why the run that made it fired. `variant` is a version trained on the same
4332
- * conversations as the one before it with one setting changed; every other
4333
- * value is a version that learned from newly reviewed conversations (or a
4334
- * member's run now).
4335
- */
4336
- trigger: TrainingTrigger;
4337
- /** See TrainingRun.recipe_note. */
4338
- recipe_note: string | null;
4339
- /** See TrainingRun.recipe_of_version. */
4340
- recipe_of_version: number | null;
4341
- /** See TrainingRun.recipe. */
4342
- recipe: TrainingRecipe;
4343
- finished_at: string | null;
4344
- }
4345
- /**
4346
- * A training rule seen as the series of versions it produces: which exist,
4347
- * which one serves, whether each beat the one before it, and what happens
4348
- * next. READ-ONLY: every decision stays on the route that owns it (promote,
4349
- * reject and roll back on the run; the version limit on the rule).
4350
- */
4351
- /**
4352
- * Whether a pipeline can make a recipe variant next, and when it cannot, which
4353
- * reason -- each is a different next step.
4354
- *
4355
- * - `off`: explore_recipes is off.
4356
- * - `waiting_for_v1`: on, but a variant varies a version and there is none yet.
4357
- * - `available`: on, with `recipe_variants_left` untried variants that fit.
4358
- * - `no_slots`: untried variants fit (`recipe_variants_left` > 0), but every
4359
- * version slot is made or taken by the run in flight. Raise `max_versions`
4360
- * (or, at 10, start a new pipeline) and one can run. Only when variants are
4361
- * left: a service that reports `no_slots` with `recipe_variants_left` of 0
4362
- * means `all_tried`, and raising the limit starts nothing.
4363
- * - `all_tried`: every variant that fits has been tried on the conversations
4364
- * it has now, whether or not a version slot is free. Raising `max_versions`
4365
- * does not start one; new reviews bring the next version.
4366
- * - `none_fit`: no variant fits this rule's training settings.
4367
- */
4368
- export type PipelineRecipeExploration = 'off' | 'waiting_for_v1' | 'available' | 'no_slots' | 'all_tried' | 'none_fit';
4369
- export interface Pipeline {
4370
- rule_id: string;
4371
- rule_name: string;
4372
- enabled: boolean;
4373
- /** The limit the member set, 1 to 10. */
4374
- max_versions: number;
4375
- /** Versions made so far. At `max_versions` the pipeline stops. */
4376
- versions_made: number;
4377
- /** The pinned base every version is a fresh adapter over, which is what makes their scores comparable. */
4378
- base_model_id: string;
4379
- base_model_revision: string;
4380
- train_type: string;
4381
- /** Null when what serves is not a version of this pipeline. */
4382
- champion_version: number | null;
4383
- /** The name the app calls. It does not change when a version is promoted. */
4384
- serving_name: string;
4385
- /**
4386
- * Another pipeline that has since promoted onto the same serving name. Only
4387
- * the most recent promotion serves; this pipeline's own champion no longer
4388
- * does. Null when nothing has taken the name over.
4389
- */
4390
- serving_taken_over_by: string | null;
4391
3984
  benchmark_id: string | null;
4392
- benchmark_name: string | null;
4393
- benchmark_decides: boolean;
4394
- benchmark_min_delta: number;
4395
- status: PipelineStatus;
4396
- /**
4397
- * The rule's own sentence about what it is waiting for or why it stopped,
4398
- * verbatim. Only as fresh as the platform's last visit: when `status` is
4399
- * `paused`, `paused_reason` and `needs_consent` are what say why.
4400
- */
4401
- next_reason: string | null;
4402
- last_checked_at: string | null;
4403
- next_due_at: string | null;
4404
- auto_promote: boolean;
4405
- /** Why the platform stopped firing this rule, or null. Same codes as the rule's. */
4406
- paused_reason: TrainingRulePausedReason | null;
4407
- /**
4408
- * The rule's terms changed and nobody has accepted them. It makes nothing
4409
- * until somebody does.
4410
- */
4411
- needs_consent: boolean;
4412
- /** The rule's explore_recipes, as it stands. */
4413
- explore_recipes: boolean;
4414
- /**
4415
- * How many recipe variants that fit this rule's settings have not yet been
4416
- * tried on the conversations the pipeline has now. Not capped by the free
4417
- * version slots; 0 while exploration is off or when no variant fits. A count
4418
- * is not a reason: read `recipe_exploration` for whether one can run next.
4419
- */
4420
- recipe_variants_left: number;
4421
- /** Whether the next recipe variant can be tried, and if not, why. */
4422
- recipe_exploration: PipelineRecipeExploration;
4423
- /** The rule's monthly limit. Null means it has none. */
4424
- monthly_ceiling_cents: number | null;
4425
- /**
4426
- * What this rule's runs have cost at most, counted toward the monthly limit
4427
- * this UTC calendar month -- the figure the limit is enforced against. An
4428
- * upper bound, not an exact spend. A run counts toward the month it was
4429
- * created in, and a run still in progress also counts toward the current
4430
- * month, at its three ceilings or what it has been billed when that is
4431
- * more. A finished run counts what it was billed, and its comparison
4432
- * machine is billed at the most it could have cost (the hourly cap for the
4433
- * time it was up, never more than the candidate ceiling). A run created in
4434
- * an earlier month that finishes in this one counts here only while it is
4435
- * still running.
4436
- */
4437
- month_spent_cents: number;
4438
- active_run: PipelineActiveRun | null;
4439
- versions: PipelineVersion[];
4440
3985
  }
4441
3986
  export interface TrainingRuleListResponse {
4442
3987
  rules: TrainingRule[];
@@ -4445,7 +3990,6 @@ export interface TrainingRuleListResponse {
4445
3990
  export interface TrainingRuleResponse {
4446
3991
  rule: TrainingRule;
4447
3992
  recent_runs: TrainingRunSummary[];
4448
- /** What this rule's runs have cost at most this UTC calendar month; see {@link Pipeline.month_spent_cents}. */
4449
3993
  month_spent_cents: number;
4450
3994
  judge_agreement: JudgeAgreement | null;
4451
3995
  }
@@ -4518,14 +4062,3 @@ export interface BenchmarkHistoryResponse {
4518
4062
  points: BenchmarkHistoryPoint[];
4519
4063
  total: number;
4520
4064
  }
4521
- export interface PipelineListResponse {
4522
- /** The newest pipelines first, at most the service's page (100). */
4523
- pipelines: Pipeline[];
4524
- /** Every pipeline in the workspace, including any not returned. */
4525
- total: number;
4526
- /** More pipelines exist than were returned; the rest are reached through `listTrainingRules`. */
4527
- truncated: boolean;
4528
- }
4529
- export interface PipelineResponse {
4530
- pipeline: Pipeline;
4531
- }
package/dist/types.js CHANGED
@@ -12,6 +12,3 @@ export const TERMINAL_RUN_STATES = [
12
12
  'superseded',
13
13
  'rolled_back',
14
14
  ];
15
- /* ── Pipelines ────────────────────────────────────────────────────────────── */
16
- /** The most versions a pipeline may be set to make (loop-service maxVersionsCeiling). */
17
- export const PIPELINE_MAX_VERSIONS_CEILING = 10;
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "runbios-sdk",
3
- "version": "0.2.13-dev.243",
3
+ "version": "0.2.13-rc.239",
4
4
  "description": "Official TypeScript SDK for the Run BiOS training and deployment platform API",
5
5
  "type": "module",
6
6
  "main": "./dist/index.js",