runbios-sdk 0.2.13-dev.242 → 0.2.13-dev.243

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/dist/index.d.ts CHANGED
@@ -36,7 +36,7 @@ import { Loop } from './resources/loop.js';
36
36
  * SDK version. Sent as part of the User-Agent header.
37
37
  * Must match package.json "version" -- enforced by a contract test.
38
38
  */
39
- export declare const VERSION = "0.2.13-dev.242";
39
+ export declare const VERSION = "0.2.13-dev.243";
40
40
  export declare class RunBiOS {
41
41
  /** Search models, fetch configs, check adapter compatibility. */
42
42
  readonly models: Models;
@@ -75,6 +75,8 @@ export { Training } from './resources/training.js';
75
75
  export { Wallet } from './resources/wallet.js';
76
76
  export { GPU, type GPURecommendation } from './resources/gpu.js';
77
77
  export { Inference, validateChatRequest, parseSSE, buildInferenceRequest, contextSizingBasis, type ContextSizingBasis, CONTEXT_DEFAULT_CEILING, CONTEXT_EDITABLE_FLOOR, type ChatMessage, type FunctionTool, type ChatCompletionParams, type ChatCompletionResponse, type ChatCompletionChunk, } from './resources/inference.js';
78
- export type { BiOSConfig, PaginatedResponse, ApiErrorBody, AvailableGpuAlternative, CapacityMinimumRequirement, InferenceBookingAccepted, Model, ModelSearchParams, ModelSearchResponse, ModelDetailResponse, ModelConfig, Dataset, DatasetListParams, DatasetListResponse, DatasetUploadParams, DatasetPreview, DatasetPreviewParams, DatasetImportHFParams, DatasetRegisterHFParams, DatasetHubSearchParams, DatasetHubPreviewParams, DatasetValidation, DatasetFormatVariant, DatasetFormatSpec, DatasetFormatSpecs, DatasetStorageUsage, TrainingMethod, RLHFAlgorithm, AdapterType, TrainingJobStatus, TrainingStatusFilter, TrainingCreateParams, TrainingListParams, TrainingListResponse, TrainingJob, TrainingMetrics, TrainingResourceMetrics, TrainingDeviceMetrics, MetricPoint, MetricGraphConfig, TrainingCheckpoint, TrainingLogs, TrainingLogEntry, TrainingStopResponse, TrainingResumeResponse, CanonicalTrainingRequest, TrainingPreflightDataset, TrainingPreflightWarning, TrainingPreflightResponse, TrainingCapabilityChoice, TrainingConfigFieldCapability, TrainingCapabilities, GPUChoice, WalletBalance, Transaction, TransactionListResponse, TransactionListParams, GPUInfo, GPUPricingResponse, GPUOptionsParams, TrainingRecommendParams, TrainingAdvisorVerdict, TrainingAdvisorRecommendation, TrainingAdvisorUserShape, TrainingAdvisorJustification, TrainingAdvisorBasis, GPUOption, GPUOptionSuggestion, GPUOptionsResponse, InferenceStatus, InferenceCreateParams, InferenceUpdateParams, InferenceDeployment, InferenceDeploymentSummary, InferenceCreateResponse, InferenceListResponse, InferenceUpdateResponse, InferenceLifecycleResponse, InferenceDeleteResponse, InferenceGPUOptionMarket, InferenceGPUOption, InferenceGPUAlternative, InferenceGPUOptionsParams, InferenceGPUOptionsResponse, InferenceCanonicalRequest, InferencePreflightWarning, InferencePreflightResponse, AdapterCompatibility, AdapterCompatibilityResponse, AdapterCompatibilityParams, SupportedArchitecture, ArchitectureScope, SupportedArchitecturesResponse, ApiKey, ApiKeyScope, ApiKeyIntrospection, Organization, OrgMember, OrgInvite, Workspace, Integration, IntegrationCreateParams, StorageUsage, StorageObject, PromptTokensDetails, ChatCompletionUsage, TrainingCadence, TrainingCombinator, ServingKind, TrainingRulePausedReason, LoopTrainingMethod, TrainType, GradersScope, TrainingTrigger, TrainingRunState, TrainingVerdict, TrainingDecision, NotifyState, EvaluationStatus, EvaluationItemStatus, EvaluationWinner, EvaluationWarningCode, ConsentVia, TrainingToolCall, TrainingMessage, TrainingRuleServing, TrainingGPURung, TrainingRule, TrainingRuleConsent, TrainingRulePreflightRefusal, TrainingRuleKeyRef, TrainingRuleKeyCheck, TrainingRulePreflight, TrainingRuleBuildSpec, TrainingRuleTriggerInput, TrainingRuleTrainingInput, TrainingRuleDeployInput, TrainingRuleMoneyInput, TrainingRuleEvaluationInput, TrainingRulePromotionInput, TrainingRuleAcceptTerms, TrainingRulePreflightRequest, TrainingRuleCreateRequest, TrainingRuleUpdateRequest, TrainingRuleConsentRequest, TrainingRuleListParams, TrainingRunPromoteRequest, TrainingRunRejectRequest, TrainingRunRollbackRequest, TrainingRunCancelRequest, TrainingRunListParams, TrainingRun, TrainingRunSummary, TrainingRunEvent, TrainingRunLinks, TrainingRunActions, EvaluationRef, EvaluationDecoding, EvaluationJudgeDimension, EvaluationDimensionScore, EvaluationGraderScore, EvaluationWarning, EvaluationMargin, JudgeAgreementDimension, JudgeAgreement, Evaluation, EvaluationItemSide, EvaluationItem, EvaluationItemListParams, JudgeAgreementParams, AgentSettings, AgentSettingsRequest, BenchmarkStatus, BenchmarkSourceKind, BenchmarkRunStatus, BenchmarkScoring, BenchmarkDimensionScore, Benchmark, BenchmarkItem, BenchmarkRun, BenchmarkHistoryPoint, BenchmarkSource, BenchmarkCreateParams, BenchmarkListParams, BenchmarkItemListParams, BenchmarkHistoryParams, BenchmarkRetireParams, TrainingRuleBenchmarkRequest, InferenceAlias, InferenceAliasRequest, TrainingRuleListResponse, TrainingRuleResponse, TrainingRuleMutationResponse, TrainingRuleDeleteResponse, TrainingRunListResponse, TrainingRunResponse, TrainingRunActionResponse, EvaluationItemsResponse, AgentSettingsResponse, InferenceAliasListResponse, InferenceAliasResponse, InferenceAliasDeleteResponse, BenchmarkListResponse, BenchmarkItemsResponse, BenchmarkHistoryResponse, } from './types.js';
78
+ export type { BiOSConfig, PaginatedResponse, ApiErrorBody, AvailableGpuAlternative, CapacityMinimumRequirement, InferenceBookingAccepted, Model, ModelSearchParams, ModelSearchResponse, ModelDetailResponse, ModelConfig, Dataset, DatasetListParams, DatasetListResponse, DatasetUploadParams, DatasetPreview, DatasetPreviewParams, DatasetImportHFParams, DatasetRegisterHFParams, DatasetHubSearchParams, DatasetHubPreviewParams, DatasetValidation, DatasetFormatVariant, DatasetFormatSpec, DatasetFormatSpecs, DatasetStorageUsage, TrainingMethod, RLHFAlgorithm, AdapterType, TrainingJobStatus, TrainingStatusFilter, TrainingCreateParams, TrainingListParams, TrainingListResponse, TrainingJob, TrainingMetrics, TrainingResourceMetrics, TrainingDeviceMetrics, MetricPoint, MetricGraphConfig, TrainingCheckpoint, TrainingLogs, TrainingLogEntry, TrainingStopResponse, TrainingResumeResponse, CanonicalTrainingRequest, TrainingPreflightDataset, TrainingPreflightWarning, TrainingPreflightResponse, TrainingCapabilityChoice, TrainingConfigFieldCapability, TrainingCapabilities, GPUChoice, WalletBalance, Transaction, TransactionListResponse, TransactionListParams, GPUInfo, GPUPricingResponse, GPUOptionsParams, TrainingRecommendParams, TrainingAdvisorVerdict, TrainingAdvisorRecommendation, TrainingAdvisorUserShape, TrainingAdvisorJustification, TrainingAdvisorBasis, GPUOption, GPUOptionSuggestion, GPUOptionsResponse, InferenceStatus, InferenceCreateParams, InferenceUpdateParams, InferenceDeployment, InferenceDeploymentSummary, InferenceCreateResponse, InferenceListResponse, InferenceUpdateResponse, InferenceLifecycleResponse, InferenceDeleteResponse, InferenceGPUOptionMarket, InferenceGPUOption, InferenceGPUAlternative, InferenceGPUOptionsParams, InferenceGPUOptionsResponse, InferenceCanonicalRequest, InferencePreflightWarning, InferencePreflightResponse, AdapterCompatibility, AdapterCompatibilityResponse, AdapterCompatibilityParams, SupportedArchitecture, ArchitectureScope, SupportedArchitecturesResponse, ApiKey, ApiKeyScope, ApiKeyIntrospection, Organization, OrgMember, OrgInvite, Workspace, Integration, IntegrationCreateParams, StorageUsage, StorageObject, PromptTokensDetails, ChatCompletionUsage, TrainingCadence, TrainingCombinator, ServingKind, TrainingRulePausedReason, LoopTrainingMethod, TrainType, GradersScope, TrainingTrigger, TrainingRecipe, TrainingRunState, TrainingVerdict, TrainingDecision, NotifyState, EvaluationStatus, EvaluationItemStatus, EvaluationWinner, EvaluationWarningCode, ConsentVia, TrainingToolCall, TrainingMessage, TrainingRuleServing, TrainingGPURung, TrainingRule, TrainingRuleConsent, TrainingRulePreflightRefusal, TrainingRuleKeyRef, TrainingRuleKeyCheck, TrainingRulePreflight, TrainingRuleBuildSpec, TrainingRuleTriggerInput, TrainingRuleTrainingInput, TrainingRuleDeployInput, TrainingRuleMoneyInput, TrainingRuleEvaluationInput, TrainingRulePromotionInput, TrainingRuleAcceptTerms, TrainingRulePreflightRequest, TrainingRuleCreateRequest, TrainingRuleUpdateRequest, TrainingRuleConsentRequest, TrainingRuleListParams, TrainingRunPromoteRequest, TrainingRunRejectRequest, TrainingRunRollbackRequest, TrainingRunCancelRequest, TrainingRunListParams, TrainingRun, TrainingRunSummary, TrainingRunEvent, TrainingRunLinks, TrainingRunActions, EvaluationRef, EvaluationDecoding, EvaluationJudgeDimension, EvaluationDimensionScore, EvaluationGraderScore, EvaluationWarning, EvaluationMargin, JudgeAgreementDimension, JudgeAgreement, Evaluation, EvaluationItemSide, EvaluationItem, EvaluationItemListParams, JudgeAgreementParams, AgentSettings, AgentSettingsRequest, BenchmarkStatus, BenchmarkSourceKind, BenchmarkRunStatus, BenchmarkScoring, BenchmarkDimensionScore, Benchmark, BenchmarkItem, BenchmarkRun, BenchmarkHistoryPoint, BenchmarkSource, BenchmarkCreateParams, BenchmarkListParams, BenchmarkItemListParams, BenchmarkHistoryParams, BenchmarkRetireParams, TrainingRuleBenchmarkRequest, InferenceAlias, InferenceAliasRequest, TrainingRuleListResponse, TrainingRuleResponse, TrainingRuleMutationResponse, TrainingRuleDeleteResponse, TrainingRunListResponse, TrainingRunResponse, TrainingRunActionResponse, EvaluationItemsResponse, AgentSettingsResponse, InferenceAliasListResponse, InferenceAliasResponse, InferenceAliasDeleteResponse, BenchmarkListResponse, BenchmarkItemsResponse, BenchmarkHistoryResponse, TrainingRuleRunRefusalCode, TrainingRuleMonthlyLimitRefusal, LoopDatasetDeleteRefusalCode, PipelineStatus, PipelineRecipeExploration, PipelineActiveRun, PipelineVersion, Pipeline, PipelineListResponse, PipelineResponse, } from './types.js';
79
79
  /** The closed set of run states a training run never leaves. */
80
80
  export { TERMINAL_RUN_STATES } from './types.js';
81
+ /** The most versions a pipeline may be set to make. */
82
+ export { PIPELINE_MAX_VERSIONS_CEILING } from './types.js';
package/dist/index.js CHANGED
@@ -36,7 +36,7 @@ import { Loop } from './resources/loop.js';
36
36
  * SDK version. Sent as part of the User-Agent header.
37
37
  * Must match package.json "version" -- enforced by a contract test.
38
38
  */
39
- export const VERSION = '0.2.13-dev.242';
39
+ export const VERSION = '0.2.13-dev.243';
40
40
  export class RunBiOS {
41
41
  /** Search models, fetch configs, check adapter compatibility. */
42
42
  models;
@@ -105,3 +105,5 @@ export { GPU } from './resources/gpu.js';
105
105
  export { Inference, validateChatRequest, parseSSE, buildInferenceRequest, contextSizingBasis, CONTEXT_DEFAULT_CEILING, CONTEXT_EDITABLE_FLOOR, } from './resources/inference.js';
106
106
  /** The closed set of run states a training run never leaves. */
107
107
  export { TERMINAL_RUN_STATES } from './types.js';
108
+ /** The most versions a pipeline may be set to make. */
109
+ export { PIPELINE_MAX_VERSIONS_CEILING } from './types.js';
@@ -1,5 +1,5 @@
1
1
  import type { HttpClient } from '../client.js';
2
- import type { LoopTrace, LoopTraceListResponse, LoopTraceListParams, LoopCaptureParams, LoopCaptureResult, LoopImportParams, LoopImportResult, LoopSignal, LoopSignalParams, LoopCandidate, LoopCandidateParams, LoopGrader, LoopGraderParams, LoopLabel, LoopLabelCount, LoopGradeResult, LoopDataset, LoopDatasetListParams, LoopDatasetItem, LoopDatasetCreateParams, LoopConfig, LoopConfigParams, LoopBuildRule, LoopBuildRuleParams, LoopBuildRuleUpdateParams, LoopStats, LoopJudge, LoopJudgeParams, LoopAgentCredential, LoopAgentStatus, LoopSampleRun, LoopSampleRunParams, LoopJudgeRun, LoopJudgeRunItems, LoopJudgeRunItemStatus, LoopJudgeWork, LoopJudgeVerdict, LoopJudgeVerdictResult, TrainingRule, TrainingRulePreflight, TrainingRulePreflightRequest, TrainingRuleCreateRequest, TrainingRuleUpdateRequest, TrainingRuleConsentRequest, TrainingRuleListParams, TrainingRuleListResponse, TrainingRuleResponse, TrainingRuleMutationResponse, TrainingRuleDeleteResponse, TrainingRun, TrainingRunListParams, TrainingRunListResponse, TrainingRunResponse, TrainingRunActionResponse, TrainingRunPromoteRequest, TrainingRunRejectRequest, TrainingRunRollbackRequest, TrainingRunCancelRequest, Evaluation, EvaluationItemListParams, EvaluationItemsResponse, JudgeAgreement, JudgeAgreementParams, Benchmark, BenchmarkCreateParams, BenchmarkHistoryParams, BenchmarkHistoryResponse, BenchmarkItemListParams, BenchmarkItemsResponse, BenchmarkListParams, BenchmarkListResponse, BenchmarkRetireParams, BenchmarkRun, TrainingRuleBenchmarkRequest, AgentSettings, AgentSettingsRequest } from '../types.js';
2
+ import type { LoopTrace, LoopTraceListResponse, LoopTraceListParams, LoopCaptureParams, LoopCaptureResult, LoopImportParams, LoopImportResult, LoopSignal, LoopSignalParams, LoopCandidate, LoopCandidateParams, LoopGrader, LoopGraderParams, LoopLabel, LoopLabelCount, LoopGradeResult, LoopDataset, LoopDatasetListParams, LoopDatasetItem, LoopDatasetCreateParams, LoopConfig, LoopConfigParams, LoopBuildRule, LoopBuildRuleParams, LoopBuildRuleUpdateParams, LoopStats, LoopJudge, LoopJudgeParams, LoopAgentCredential, LoopAgentStatus, LoopSampleRun, LoopSampleRunParams, LoopJudgeRun, LoopJudgeRunItems, LoopJudgeRunItemStatus, LoopJudgeWork, LoopJudgeVerdict, LoopJudgeVerdictResult, TrainingRule, TrainingRulePreflight, TrainingRulePreflightRequest, TrainingRuleCreateRequest, TrainingRuleUpdateRequest, TrainingRuleConsentRequest, TrainingRuleListParams, TrainingRuleListResponse, TrainingRuleResponse, TrainingRuleMutationResponse, TrainingRuleDeleteResponse, TrainingRun, TrainingRunListParams, TrainingRunListResponse, TrainingRunResponse, TrainingRunActionResponse, TrainingRunPromoteRequest, TrainingRunRejectRequest, TrainingRunRollbackRequest, TrainingRunCancelRequest, Evaluation, EvaluationItemListParams, EvaluationItemsResponse, JudgeAgreement, JudgeAgreementParams, Benchmark, BenchmarkCreateParams, BenchmarkHistoryParams, BenchmarkHistoryResponse, BenchmarkItemListParams, BenchmarkItemsResponse, BenchmarkListParams, BenchmarkListResponse, BenchmarkRetireParams, BenchmarkRun, TrainingRuleBenchmarkRequest, AgentSettings, AgentSettingsRequest, Pipeline, PipelineListResponse } from '../types.js';
3
3
  /**
4
4
  * The Conscious Loop -- capture what your model was asked and answered, record
5
5
  * whether it was right, and turn those judgements into training data.
@@ -335,7 +335,10 @@ export declare class Loop {
335
335
  * Delete a training set.
336
336
  *
337
337
  * The conversations it was built from are untouched -- a set is a selection,
338
- * and discarding the selection must not discard the evidence.
338
+ * and discarding the selection must not discard the evidence. A set a
339
+ * training run was trained on is refused `409 DATASET_IN_USE` and kept, as
340
+ * the record of what that run learned from. See
341
+ * `LoopDatasetDeleteRefusalCode`.
339
342
  */
340
343
  deleteDataset(id: string): Promise<{
341
344
  deleted: boolean;
@@ -614,11 +617,12 @@ export declare class Loop {
614
617
  */
615
618
  listTrainingRules(params?: TrainingRuleListParams): Promise<TrainingRuleListResponse>;
616
619
  /**
617
- * Read one rule with its recent runs, what it has spent this month, and how
618
- * far its judge agrees with your own reviewers.
620
+ * Read one rule with its recent runs, what its runs this month have cost at
621
+ * most, and how far its judge agrees with your own reviewers.
619
622
  *
620
623
  * Returned whole rather than unwrapped to the rule: `month_spent_cents` is
621
- * the number that says whether the monthly ceiling is about to stop it.
624
+ * the number that says whether the monthly ceiling is about to stop it. It
625
+ * is an upper bound, not an exact spend: see `Pipeline.month_spent_cents`.
622
626
  */
623
627
  getTrainingRule(id: string): Promise<TrainingRuleResponse>;
624
628
  /**
@@ -626,8 +630,11 @@ export declare class Loop {
626
630
  * that can be cleared.
627
631
  *
628
632
  * A money-bearing change -- a ceiling, a model, a GPU ladder, the promotion
629
- * policy -- bumps `revision`, clears the recorded consent and STOPS the rule
630
- * firing until someone accepts the new amounts. The reply says so in
633
+ * policy, or changing `explore_recipes` in EITHER direction -- bumps
634
+ * `revision`, clears the recorded consent and STOPS the rule firing until
635
+ * someone accepts the new amounts. Turning recipe variants OFF does this
636
+ * too: the pipeline then makes no version at all, variant or not, until the
637
+ * terms are accepted again. The reply says so in
631
638
  * `consent_required`, and carries a fresh `preflight` with the new figures.
632
639
  * Pass `expected_revision` to be refused with `REVISION_MISMATCH` rather
633
640
  * than overwrite an edit somebody else made in the meantime.
@@ -654,8 +661,17 @@ export declare class Loop {
654
661
  * Fire a rule now, without waiting for its cadence.
655
662
  *
656
663
  * Bypasses the schedule and `min_new_rows` only. The row floors, the money
657
- * ceilings and the consent all still apply, so this can answer `409
658
- * CONSENT_REQUIRED` or `422 NOT_ENOUGH_ROWS` with the counts it needed.
664
+ * ceilings, the consent, the version limit and the monthly limit all still
665
+ * apply, so this can answer `409 CONSENT_REQUIRED`, `409 RULE_PAUSED`,
666
+ * `409 RUN_ACTIVE`, `409 VERSION_LIMIT_REACHED` (the pipeline has made
667
+ * `max_versions` versions), `409 MONTHLY_LIMIT_REACHED` (the most this
668
+ * month's runs can have cost plus the most one run may cost would pass the
669
+ * monthly limit; the error
670
+ * body carries `month_spent_cents`, `monthly_ceiling_cents`, `run_max_cents`
671
+ * and `resumes_at`), `409 AGENT_OFF` (the workspace's Conscious Loop agent
672
+ * is not turned on, so a trained model could not be compared) or
673
+ * `422 NOT_ENOUGH_ROWS` with the counts it needed. See
674
+ * `TrainingRuleRunRefusalCode`.
659
675
  */
660
676
  runTrainingRule(id: string): Promise<TrainingRun>;
661
677
  /**
@@ -710,6 +726,34 @@ export declare class Loop {
710
726
  * cancelling a training job does not refund the hours it burned.
711
727
  */
712
728
  cancelTrainingRun(id: string, params?: TrainingRunCancelRequest): Promise<TrainingRunActionResponse>;
729
+ /**
730
+ * Every training rule in the workspace, seen as the series of versions it
731
+ * produced: which exist, which one serves (`champion_version`), how each did
732
+ * against the champion of its day and on the standing benchmark, the run in
733
+ * flight and which version it will be, and what the pipeline is waiting for.
734
+ *
735
+ * Read-only. Every decision stays on the route that owns it -- promote,
736
+ * reject and roll back on the run, the version limit and `explore_recipes`
737
+ * on the rule.
738
+ *
739
+ * RETURNED WITH ITS ENVELOPE, because the list is not always all of them:
740
+ * the service returns the newest 100, `total` is every pipeline in the
741
+ * workspace and `truncated` is true when more exist than were returned. A
742
+ * caller that shows `pipelines` alone presents a short list as the whole of
743
+ * it. The rest are reached through `listTrainingRules`, which pages.
744
+ */
745
+ listPipelines(): Promise<PipelineListResponse>;
746
+ /**
747
+ * One pipeline, by its training rule's id.
748
+ *
749
+ * `status` says whether it is the platform working (`running`), the member
750
+ * who has to act (`needs_review`, `needs_funds`), or nothing at all until
751
+ * somebody does (`paused`, with `paused_reason` or `needs_consent` saying
752
+ * why; `complete` at `max_versions`). `month_spent_cents` is the figure the
753
+ * monthly limit is enforced against: what runs started this month have cost
754
+ * at most, not an exact spend.
755
+ */
756
+ getPipeline(id: string): Promise<Pipeline>;
713
757
  /**
714
758
  * The comparison behind a verdict: both models on the same held-out rows,
715
759
  * with identical decoding, judge and grader names resolved.
@@ -440,7 +440,10 @@ export class Loop {
440
440
  * Delete a training set.
441
441
  *
442
442
  * The conversations it was built from are untouched -- a set is a selection,
443
- * and discarding the selection must not discard the evidence.
443
+ * and discarding the selection must not discard the evidence. A set a
444
+ * training run was trained on is refused `409 DATASET_IN_USE` and kept, as
445
+ * the record of what that run learned from. See
446
+ * `LoopDatasetDeleteRefusalCode`.
444
447
  */
445
448
  async deleteDataset(id) {
446
449
  return this._http.fetchDelete(`/api/loop/datasets/${encodeURIComponent(id)}`);
@@ -846,11 +849,12 @@ export class Loop {
846
849
  return this._http.fetchGet(`/api/loop/training-rules${qs ? `?${qs}` : ''}`);
847
850
  }
848
851
  /**
849
- * Read one rule with its recent runs, what it has spent this month, and how
850
- * far its judge agrees with your own reviewers.
852
+ * Read one rule with its recent runs, what its runs this month have cost at
853
+ * most, and how far its judge agrees with your own reviewers.
851
854
  *
852
855
  * Returned whole rather than unwrapped to the rule: `month_spent_cents` is
853
- * the number that says whether the monthly ceiling is about to stop it.
856
+ * the number that says whether the monthly ceiling is about to stop it. It
857
+ * is an upper bound, not an exact spend: see `Pipeline.month_spent_cents`.
854
858
  */
855
859
  async getTrainingRule(id) {
856
860
  return this._http.fetchGet(`/api/loop/training-rules/${encodeURIComponent(id)}`);
@@ -860,8 +864,11 @@ export class Loop {
860
864
  * that can be cleared.
861
865
  *
862
866
  * A money-bearing change -- a ceiling, a model, a GPU ladder, the promotion
863
- * policy -- bumps `revision`, clears the recorded consent and STOPS the rule
864
- * firing until someone accepts the new amounts. The reply says so in
867
+ * policy, or changing `explore_recipes` in EITHER direction -- bumps
868
+ * `revision`, clears the recorded consent and STOPS the rule firing until
869
+ * someone accepts the new amounts. Turning recipe variants OFF does this
870
+ * too: the pipeline then makes no version at all, variant or not, until the
871
+ * terms are accepted again. The reply says so in
865
872
  * `consent_required`, and carries a fresh `preflight` with the new figures.
866
873
  * Pass `expected_revision` to be refused with `REVISION_MISMATCH` rather
867
874
  * than overwrite an edit somebody else made in the meantime.
@@ -893,8 +900,17 @@ export class Loop {
893
900
  * Fire a rule now, without waiting for its cadence.
894
901
  *
895
902
  * Bypasses the schedule and `min_new_rows` only. The row floors, the money
896
- * ceilings and the consent all still apply, so this can answer `409
897
- * CONSENT_REQUIRED` or `422 NOT_ENOUGH_ROWS` with the counts it needed.
903
+ * ceilings, the consent, the version limit and the monthly limit all still
904
+ * apply, so this can answer `409 CONSENT_REQUIRED`, `409 RULE_PAUSED`,
905
+ * `409 RUN_ACTIVE`, `409 VERSION_LIMIT_REACHED` (the pipeline has made
906
+ * `max_versions` versions), `409 MONTHLY_LIMIT_REACHED` (the most this
907
+ * month's runs can have cost plus the most one run may cost would pass the
908
+ * monthly limit; the error
909
+ * body carries `month_spent_cents`, `monthly_ceiling_cents`, `run_max_cents`
910
+ * and `resumes_at`), `409 AGENT_OFF` (the workspace's Conscious Loop agent
911
+ * is not turned on, so a trained model could not be compared) or
912
+ * `422 NOT_ENOUGH_ROWS` with the counts it needed. See
913
+ * `TrainingRuleRunRefusalCode`.
898
914
  */
899
915
  async runTrainingRule(id) {
900
916
  const res = await this._http.fetchPost(`/api/loop/training-rules/${encodeURIComponent(id)}/run`, {});
@@ -980,6 +996,47 @@ export class Loop {
980
996
  async cancelTrainingRun(id, params = {}) {
981
997
  return this._http.fetchPost(`/api/loop/training-runs/${encodeURIComponent(id)}/cancel`, params);
982
998
  }
999
+ // ── pipelines ─────────────────────────────────────────────────────────
1000
+ /**
1001
+ * Every training rule in the workspace, seen as the series of versions it
1002
+ * produced: which exist, which one serves (`champion_version`), how each did
1003
+ * against the champion of its day and on the standing benchmark, the run in
1004
+ * flight and which version it will be, and what the pipeline is waiting for.
1005
+ *
1006
+ * Read-only. Every decision stays on the route that owns it -- promote,
1007
+ * reject and roll back on the run, the version limit and `explore_recipes`
1008
+ * on the rule.
1009
+ *
1010
+ * RETURNED WITH ITS ENVELOPE, because the list is not always all of them:
1011
+ * the service returns the newest 100, `total` is every pipeline in the
1012
+ * workspace and `truncated` is true when more exist than were returned. A
1013
+ * caller that shows `pipelines` alone presents a short list as the whole of
1014
+ * it. The rest are reached through `listTrainingRules`, which pages.
1015
+ */
1016
+ async listPipelines() {
1017
+ const res = await this._http.fetchGet('/api/loop/pipelines');
1018
+ const pipelines = res.pipelines || [];
1019
+ return {
1020
+ pipelines,
1021
+ // An older service sends neither: its list is taken at its word.
1022
+ total: typeof res.total === 'number' ? res.total : pipelines.length,
1023
+ truncated: res.truncated === true,
1024
+ };
1025
+ }
1026
+ /**
1027
+ * One pipeline, by its training rule's id.
1028
+ *
1029
+ * `status` says whether it is the platform working (`running`), the member
1030
+ * who has to act (`needs_review`, `needs_funds`), or nothing at all until
1031
+ * somebody does (`paused`, with `paused_reason` or `needs_consent` saying
1032
+ * why; `complete` at `max_versions`). `month_spent_cents` is the figure the
1033
+ * monthly limit is enforced against: what runs started this month have cost
1034
+ * at most, not an exact spend.
1035
+ */
1036
+ async getPipeline(id) {
1037
+ const res = await this._http.fetchGet(`/api/loop/pipelines/${encodeURIComponent(id)}`);
1038
+ return res.pipeline;
1039
+ }
983
1040
  // ── the comparison report ─────────────────────────────────────────────
984
1041
  /**
985
1042
  * The comparison behind a verdict: both models on the same held-out rows,
package/dist/types.d.ts CHANGED
@@ -2533,6 +2533,28 @@ export interface LoopSignalParams {
2533
2533
  reason?: string;
2534
2534
  author?: string;
2535
2535
  metadata?: Record<string, unknown>;
2536
+ /**
2537
+ * Which training pipeline this feedback feeds, named in the SAME call.
2538
+ *
2539
+ * A build rule selects conversations by label and a training rule trains from
2540
+ * that build rule, so a label IS the pipeline a conversation goes down.
2541
+ * Putting one on used to need a second request after this one — and that
2542
+ * second request is the one that gets skipped, by a script that handles the
2543
+ * 201 and moves on, by a retry that succeeds here and fails there, by an
2544
+ * integration whose author never knew labelling was a step.
2545
+ *
2546
+ * What it leaves behind is a reviewed conversation in no pipeline: counted in
2547
+ * every "reviewed" total and selected by nothing.
2548
+ *
2549
+ * Written in one transaction with the verdict: either both land or neither
2550
+ * does. Omit them and nothing changes — the overwhelming majority of feedback
2551
+ * names no pipeline and behaves exactly as it always has.
2552
+ */
2553
+ labels?: string[];
2554
+ /** The same, by dimension: `{ category: 'billing' }`. */
2555
+ attributes?: Record<string, string>;
2556
+ /** The master group the bare labels belong to. */
2557
+ parent?: string;
2536
2558
  }
2537
2559
  export interface LoopTrace {
2538
2560
  id: string;
@@ -3125,9 +3147,46 @@ export type TrainingRulePausedReason = 'consent_invalid' | 'insufficient_funds'
3125
3147
  * republished to keep up with a server-side capability flag.
3126
3148
  */
3127
3149
  export type LoopTrainingMethod = 'sft' | 'rlhf';
3150
+ /**
3151
+ * What a rule's stored `train_type` may READ as.
3152
+ *
3153
+ * Still three, and deliberately: the column has always allowed `full`, so a
3154
+ * rule created before standing rules were narrowed to adapters can still come
3155
+ * back carrying it. Narrowing the response type would make this SDK
3156
+ * misrepresent a row that really exists.
3157
+ */
3128
3158
  export type TrainType = 'lora' | 'qlora' | 'full';
3159
+ /**
3160
+ * What a rule may be SET to, which is narrower.
3161
+ *
3162
+ * loop-service refuses `full` on preflight, create and update for a standing
3163
+ * rule: a rule trains unattended and on repeat, and a full fine-tune rewrites
3164
+ * every weight instead of adding a small adapter, so it cannot be compared or
3165
+ * rolled back cheaply. Typing the request as the wider set handed callers a
3166
+ * value guaranteed to come back a 400. One-off full fine-tunes are unaffected
3167
+ * — they are a different API.
3168
+ */
3169
+ export type RuleTrainType = 'lora' | 'qlora';
3129
3170
  export type GradersScope = 'source' | 'all' | 'none';
3130
- export type TrainingTrigger = 'rows' | 'cadence' | 'both' | 'manual';
3171
+ /**
3172
+ * Why a run fired. `variant` is a recipe variant: the pipeline had nothing new
3173
+ * to learn, so it trained the same base model on the same conversations with
3174
+ * one training setting changed (see {@link TrainingRecipe}).
3175
+ */
3176
+ export type TrainingTrigger = 'rows' | 'cadence' | 'both' | 'manual' | 'variant';
3177
+ /**
3178
+ * The four training settings a recipe variant may change, as the run was
3179
+ * actually trained: the training service's own defaults are filled in where
3180
+ * the rule left a knob unset, so two recipes can be compared value for value.
3181
+ * A knob that does not apply to the run (a LoRA rank on a full fine-tune) is
3182
+ * null, never a guessed number.
3183
+ */
3184
+ export interface TrainingRecipe {
3185
+ learning_rate: number | null;
3186
+ num_train_epochs: number | null;
3187
+ lora_rank: number | null;
3188
+ lora_alpha: number | null;
3189
+ }
3131
3190
  export type TrainingRunState = 'built' | 'registering' | 'dataset_validating' | 'preflighting' | 'job_creating' | 'training' | 'checkpoint_ready' | 'candidate_booking' | 'candidate_running' | 'evaluating' | 'reported' | 'awaiting_review' | 'promoting' | 'waiting_funds' | 'promoted' | 'rejected' | 'expired' | 'failed' | 'budget_stopped' | 'cancelled' | 'superseded' | 'rolled_back';
3132
3191
  /** The closed set a run never leaves. */
3133
3192
  export declare const TERMINAL_RUN_STATES: readonly TrainingRunState[];
@@ -3218,6 +3277,30 @@ export interface TrainingRule {
3218
3277
  eval_max_rows: number;
3219
3278
  eval_max_tokens: number;
3220
3279
  min_holdout_rows: number;
3280
+ /**
3281
+ * How many versions this pipeline may make, 1 to 10 (five unless changed).
3282
+ * A version is a trained model whose comparison was reported having scored
3283
+ * at least one conversation, or one a member put live; a run that failed,
3284
+ * was cancelled or compared nothing is not one and does not count. At the
3285
+ * limit the pipeline stops, and a manual run is refused with
3286
+ * `409 VERSION_LIMIT_REACHED`, until the limit is raised.
3287
+ */
3288
+ max_versions: number;
3289
+ /**
3290
+ * Try other training recipes when there is nothing new to learn.
3291
+ *
3292
+ * MONEY-BEARING. When on, a pipeline that has made at least one version, and
3293
+ * has nothing reviewed since it that adds anything new to train on (nothing
3294
+ * reviewed at all, or only reviews -- thumbs-down with no correction, say --
3295
+ * that give it no row the last version's set lacked), trains the same base
3296
+ * model on the same conversations with ONE setting changed, and that model
3297
+ * takes over only if it beats what serves the app, like any other version.
3298
+ * Each try is a full paid run inside the rule's per-run ceilings and its
3299
+ * monthly limit, so changing this in EITHER direction changes the terms: the
3300
+ * rule stops firing until the member accepts them again. False unless
3301
+ * somebody turned it on.
3302
+ */
3303
+ explore_recipes: boolean;
3221
3304
  /**
3222
3305
  * The standing benchmark replayed on every run of this rule, beside the
3223
3306
  * per-run comparison and never instead of it. Null is the ordinary state.
@@ -3227,6 +3310,32 @@ export interface TrainingRule {
3227
3310
  * its own. It raises no amount, so it does not invalidate consent.
3228
3311
  */
3229
3312
  benchmark_id: string | null;
3313
+ /**
3314
+ * Whether that benchmark DECIDES the verdict, or only reports a number.
3315
+ *
3316
+ * False for every rule that has not asked, which is the point of it being
3317
+ * separate from `benchmark_id`. Attaching a benchmark means "replay this
3318
+ * fixed set on every run and put the score on the report"; it does not mean
3319
+ * "let that score overrule the comparison this rule was built on".
3320
+ *
3321
+ * When it is on it cuts both ways. A run whose held-out split came up short
3322
+ * can be decided at all -- before this, such a run trained a model, billed a
3323
+ * GPU and could never promote -- and a candidate that wins on fresh
3324
+ * conversations while losing ground on the fixed set is refused.
3325
+ */
3326
+ benchmark_decides: boolean;
3327
+ /**
3328
+ * How many of the benchmark's pinned conversations must have scored before
3329
+ * it may decide anything. A replay that got through four of its forty has
3330
+ * not measured the new model.
3331
+ */
3332
+ benchmark_min_rows: number;
3333
+ /**
3334
+ * The bar on the replay's own candidate-minus-incumbent delta. Like-for-like
3335
+ * WITHIN one replay -- both models, same pinned rows, same frozen judge, one
3336
+ * pass -- and not comparable between runs.
3337
+ */
3338
+ benchmark_min_delta: number;
3230
3339
  auto_promote: boolean;
3231
3340
  promote_margin: number;
3232
3341
  promote_min_win_rate: number;
@@ -3307,6 +3416,14 @@ export interface TrainingRulePreflight {
3307
3416
  key_check: TrainingRuleKeyCheck;
3308
3417
  terms_text: string;
3309
3418
  terms_version: string;
3419
+ /**
3420
+ * The machines the hourly amounts are per hour of: the ladders the platform
3421
+ * chose when the request left them empty, or the caller's own echoed back
3422
+ * unchanged. An amount per hour means nothing without knowing what it is per
3423
+ * hour of, so show these beside the estimate before anyone accepts it.
3424
+ */
3425
+ train_gpu_priorities: TrainingGPURung[];
3426
+ deploy_gpu_priorities: TrainingGPURung[];
3310
3427
  }
3311
3428
  export interface TrainingRuleBuildSpec {
3312
3429
  method: string;
@@ -3340,6 +3457,18 @@ export interface TrainingRuleTriggerInput {
3340
3457
  combinator?: TrainingCombinator;
3341
3458
  /** null removes the row floor, leaving the schedule as the only trigger. */
3342
3459
  min_new_rows?: number | null;
3460
+ /** 1 to 10. Absent leaves it as it is; absent on a create means five. */
3461
+ max_versions?: number;
3462
+ /**
3463
+ * See {@link TrainingRule.explore_recipes}: every variant it allows is a
3464
+ * full paid run, so changing it -- on OR off -- is a money-bearing edit that
3465
+ * pauses the rule, and it makes no version of any kind, until the new terms
3466
+ * are accepted. Turning it off to save money stops the pipeline too, until
3467
+ * somebody accepts again. It sits beside `max_versions`
3468
+ * because both say when the pipeline makes another version. Absent leaves
3469
+ * it as it is; absent on a create means off.
3470
+ */
3471
+ explore_recipes?: boolean;
3343
3472
  }
3344
3473
  export interface TrainingRuleTrainingInput {
3345
3474
  model_id?: string;
@@ -3351,7 +3480,7 @@ export interface TrainingRuleTrainingInput {
3351
3480
  * plain supervised training means.
3352
3481
  */
3353
3482
  rlhf_type?: string | null;
3354
- train_type?: TrainType;
3483
+ train_type?: RuleTrainType;
3355
3484
  config?: Record<string, unknown>;
3356
3485
  train_gpu_priorities?: TrainingGPURung[];
3357
3486
  train_max_price_hour_cents?: number;
@@ -3434,6 +3563,19 @@ export interface TrainingRuleCreateRequest extends TrainingRulePreflightRequest
3434
3563
  * preflight body and not on the update body.
3435
3564
  */
3436
3565
  benchmark_id?: string;
3566
+ /**
3567
+ * And whether that benchmark decides, for the same reason `benchmark_id` is
3568
+ * here: run 1 reads its gate off the snapshot it freezes, and a gate applied
3569
+ * by a second call lands after run 1 has been decided without it.
3570
+ *
3571
+ * `400 BENCHMARK_GATE_HAS_NO_SET` when it is true with no benchmark,
3572
+ * `400 BENCHMARK_GATE_NEEDS_MIN_ROWS` when it is true with no floor.
3573
+ */
3574
+ benchmark_decides?: boolean;
3575
+ /** Cannot exceed the set's size: `400 BENCHMARK_GATE_UNREACHABLE`. */
3576
+ benchmark_min_rows?: number;
3577
+ /** Zero -- the default -- means "must not lose ground". */
3578
+ benchmark_min_delta?: number;
3437
3579
  }
3438
3580
  export interface TrainingRuleUpdateRequest extends TrainingRulePreflightRequest {
3439
3581
  expected_revision?: number;
@@ -3479,6 +3621,18 @@ export interface TrainingRun {
3479
3621
  authorized_by: string;
3480
3622
  rule_snapshot: Record<string, unknown>;
3481
3623
  trigger: TrainingTrigger;
3624
+ /**
3625
+ * For a recipe variant, what it changed, in plain words ("half the learning
3626
+ * rate"). Null for every run that trained on the ordinary recipe.
3627
+ */
3628
+ recipe_note: string | null;
3629
+ /**
3630
+ * The version whose recipe this run varied. Null when the variant started
3631
+ * from the rule's own settings, and for every run that is not a variant.
3632
+ */
3633
+ recipe_of_version: number | null;
3634
+ /** The four recipe knobs this run trained with, defaults filled in. */
3635
+ recipe: TrainingRecipe;
3482
3636
  fired_reason: string | null;
3483
3637
  state: TrainingRunState;
3484
3638
  state_entered_at: string;
@@ -3487,6 +3641,13 @@ export interface TrainingRun {
3487
3641
  last_reason: string | null;
3488
3642
  error_code: string | null;
3489
3643
  last_error: string | null;
3644
+ /**
3645
+ * What to DO about the failure, in one sentence, derived server-side from
3646
+ * `error_code` -- the same words the failure email carries, so the two
3647
+ * cannot drift. Null when the run has not failed, or when the failure has
3648
+ * no remedy worth printing.
3649
+ */
3650
+ error_next_step: string | null;
3490
3651
  training_ceiling_cents: number;
3491
3652
  candidate_ceiling_cents: number;
3492
3653
  eval_ceiling_cents: number;
@@ -3503,6 +3664,13 @@ export interface TrainingRun {
3503
3664
  queue_deadline_at: string | null;
3504
3665
  checkpoint_id: string | null;
3505
3666
  checkpoint_step: number | null;
3667
+ /**
3668
+ * Which version of the pipeline this run produced. Null for a run that is
3669
+ * not a version: one that failed, was cancelled, or whose comparison scored
3670
+ * nothing, and that nobody put live. `seq` counts attempts; this counts
3671
+ * models that competed, plus any a member promoted without a comparison.
3672
+ */
3673
+ version_no: number | null;
3506
3674
  training_eval_loss: number | null;
3507
3675
  candidate_deployment_id: string | null;
3508
3676
  candidate_name: string | null;
@@ -3546,10 +3714,20 @@ export interface TrainingRunSummary {
3546
3714
  workspace_id: string;
3547
3715
  seq: number;
3548
3716
  trigger: TrainingTrigger;
3717
+ /** See TrainingRun.recipe_note. On the list row too: the Runs table is where a variant is first seen. */
3718
+ recipe_note: string | null;
3719
+ /** See TrainingRun.recipe_of_version. */
3720
+ recipe_of_version: number | null;
3721
+ /** See TrainingRun.recipe. */
3722
+ recipe: TrainingRecipe;
3549
3723
  state: TrainingRunState;
3550
3724
  state_entered_at: string;
3551
3725
  last_reason: string | null;
3552
3726
  error_code: string | null;
3727
+ /** See TrainingRun.error_next_step. On the list row too: the Runs table is where a failure is first seen. */
3728
+ error_next_step: string | null;
3729
+ /** See TrainingRun.version_no. */
3730
+ version_no: number | null;
3553
3731
  verdict: TrainingVerdict | null;
3554
3732
  decision: TrainingDecision | null;
3555
3733
  train_rows: number | null;
@@ -3975,13 +4153,290 @@ export interface BenchmarkRetireParams {
3975
4153
  reason?: string;
3976
4154
  }
3977
4155
  /**
3978
- * Attach a benchmark to a rule, or detach it with a present null.
4156
+ * Attach a benchmark to a rule, detach it with a present null, or move the bar
4157
+ * it decides on.
4158
+ *
4159
+ * ABSENT IS "LEAVE IT", PRESENT IS "MAKE IT THIS", and that applies to every
4160
+ * field here. This is the route that attaches a benchmark AND the route that
4161
+ * moves its bar a month later, so a call naming only `benchmark_id` must not
4162
+ * reset a floor somebody chose, and a call naming only `benchmark_min_delta`
4163
+ * must not detach the set.
3979
4164
  *
3980
- * Not optional, and not omittable: this route sets the field, so an absent key
3981
- * would be a request with nothing in it. Send the id to attach, null to detach.
4165
+ * Detaching is the one exception: a present null `benchmark_id` clears
4166
+ * `benchmark_decides` with it, because a gate with nothing behind it is a
4167
+ * verdict waiting on a measurement that will never come. Asking for both in
4168
+ * one request -- null id and `benchmark_decides: true` -- is refused rather
4169
+ * than resolved, because either resolution is a guess about which half you
4170
+ * meant.
3982
4171
  */
3983
4172
  export interface TrainingRuleBenchmarkRequest {
4173
+ benchmark_id?: string | null;
4174
+ benchmark_decides?: boolean;
4175
+ benchmark_min_rows?: number;
4176
+ benchmark_min_delta?: number;
4177
+ }
4178
+ /**
4179
+ * The machine codes `runTrainingRule` can be refused with. Each is the
4180
+ * platform declining to start a paid run it would only refuse a minute later,
4181
+ * so each is said in front of the caller rather than left for the rule's
4182
+ * `last_reason` to explain after they have stopped looking.
4183
+ *
4184
+ * - `CONSENT_REQUIRED` (409): the terms changed and nobody accepted them.
4185
+ * - `RULE_PAUSED` (409): the rule is switched off or the platform stopped it.
4186
+ * - `RUN_ACTIVE` (409): one run at a time, so two cannot race to promote.
4187
+ * - `VERSION_LIMIT_REACHED` (409): the pipeline has made `max_versions`
4188
+ * versions. Raise the limit or start a new pipeline.
4189
+ * - `MONTHLY_LIMIT_REACHED` (409): the most this month's runs can have cost
4190
+ * plus the most one more run may cost would pass `monthly_ceiling_cents`.
4191
+ * The body carries the figures:
4192
+ * see {@link TrainingRuleMonthlyLimitRefusal}.
4193
+ * - `AGENT_OFF` (409): the workspace's Conscious Loop agent is not turned on.
4194
+ * Every comparison runs under the agent's key, so a run started without one
4195
+ * would pay for training and could not be compared. Turn it on in Agent
4196
+ * settings, then run the rule.
4197
+ * - `NOT_ENOUGH_ROWS` (422): too few reviewed or held-out conversations, with
4198
+ * `train_rows`, `holdout_rows` and the `floor` it needed.
4199
+ */
4200
+ export type TrainingRuleRunRefusalCode = 'CONSENT_REQUIRED' | 'RULE_PAUSED' | 'RUN_ACTIVE' | 'VERSION_LIMIT_REACHED' | 'MONTHLY_LIMIT_REACHED' | 'AGENT_OFF' | 'NOT_ENOUGH_ROWS';
4201
+ /**
4202
+ * The machine codes `deleteDataset` can be refused with.
4203
+ *
4204
+ * - `DATASET_IN_USE` (409): a training run was trained on this set, so it is
4205
+ * kept as the record of what that run learned from.
4206
+ */
4207
+ export type LoopDatasetDeleteRefusalCode = 'DATASET_IN_USE';
4208
+ /**
4209
+ * The body of a manual run refused `409 MONTHLY_LIMIT_REACHED`, as
4210
+ * `ApiError.body` carries it.
4211
+ *
4212
+ * The test is the worst case, not the average: a run may start only if the
4213
+ * most the rule's runs this UTC calendar month can have cost
4214
+ * (`month_spent_cents`), plus the most the run about to start may spend (its
4215
+ * three ceilings added up), fits under the limit.
4216
+ */
4217
+ export interface TrainingRuleMonthlyLimitRefusal {
4218
+ error: {
4219
+ code: 'MONTHLY_LIMIT_REACHED';
4220
+ message: string;
4221
+ };
4222
+ /** What this rule's runs have cost at most this UTC calendar month; see {@link Pipeline.month_spent_cents}. */
4223
+ month_spent_cents: number;
4224
+ /** The rule's monthly limit. */
4225
+ monthly_ceiling_cents: number;
4226
+ /** The most one run of this rule may spend: training + candidate + evaluation ceilings. */
4227
+ run_max_cents: number;
4228
+ /**
4229
+ * The first instant of the next UTC month, when the counted spend starts
4230
+ * again from nothing. When `run_max_cents` alone exceeds the limit, no month
4231
+ * will ever fit and only raising the limit helps; `message` says which.
4232
+ */
4233
+ resumes_at: string;
4234
+ }
4235
+ /** The most versions a pipeline may be set to make (loop-service maxVersionsCeiling). */
4236
+ export declare const PIPELINE_MAX_VERSIONS_CEILING = 10;
4237
+ /**
4238
+ * What a pipeline is doing. `needs_review` and `needs_funds` are a finished
4239
+ * version waiting on the MEMBER -- a decision, or a top-up -- which is not the
4240
+ * platform working on it.
4241
+ *
4242
+ * `paused` is a rule that is switched off, AND an enabled rule the platform
4243
+ * has stopped firing (`paused_reason`) or whose terms changed and were never
4244
+ * accepted (`needs_consent`). `complete` is a pipeline at its `max_versions`.
4245
+ */
4246
+ export type PipelineStatus = 'running' | 'needs_review' | 'needs_funds' | 'waiting' | 'paused' | 'complete';
4247
+ /**
4248
+ * The run in flight. It is not a version until its comparison is reported
4249
+ * having scored at least one conversation, or a member puts it live.
4250
+ */
4251
+ export interface PipelineActiveRun {
4252
+ run_id: string;
4253
+ state: TrainingRunState;
4254
+ since: string;
4255
+ /**
4256
+ * The version number it will take if it becomes one: if its comparison
4257
+ * scores something, or if a member promotes it anyway.
4258
+ */
4259
+ will_be_version: number;
4260
+ version_no: number | null;
4261
+ last_reason: string | null;
4262
+ last_error: string | null;
4263
+ train_rows: number | null;
4264
+ holdout_rows: number | null;
4265
+ /**
4266
+ * Its comparison scored at least one conversation. False on a reported run
4267
+ * means the held-back set was too small to compare anything: that run is
4268
+ * not a version, and becomes one only if a member promotes it.
4269
+ */
4270
+ competed: boolean;
4271
+ /** What recipe it is trying, when it is a recipe variant. See TrainingRun.recipe_note. */
4272
+ recipe_note: string | null;
4273
+ }
4274
+ /**
4275
+ * One version: a trained model whose comparison scored at least one
4276
+ * conversation, or one a member put live without one. The second kind has
4277
+ * `rows_scored` 0 and `win_rate` null, and may be the champion.
4278
+ */
4279
+ export interface PipelineVersion {
4280
+ version: number;
4281
+ run_id: string;
4282
+ seq: number;
4283
+ state: TrainingRunState;
4284
+ verdict: TrainingVerdict | null;
4285
+ decision: TrainingDecision | null;
4286
+ decided_at: string | null;
4287
+ /** True for the version the app is served by today. */
4288
+ is_champion: boolean;
4289
+ checkpoint_id: string | null;
4290
+ train_rows: number | null;
4291
+ holdout_rows: number | null;
4292
+ /** Head-to-head against the champion of its day, on conversations neither trained on. */
4293
+ rows_scored: number | null;
4294
+ win_rate: number | null;
4295
+ mean_delta: number | null;
4296
+ /**
4297
+ * The standing benchmark's replay of this version, when the rule has one.
4298
+ * Only scores with `benchmark_comparable` sit on one axis.
4299
+ */
4300
+ benchmark_score: number | null;
4301
+ benchmark_rows: number | null;
4302
+ benchmark_rows_total: number | null;
4303
+ benchmark_status: string | null;
4304
+ /**
4305
+ * False for a score that is not on today's yardstick: measured on an earlier
4306
+ * revision of the benchmark's items, or on a different benchmark than the
4307
+ * one attached now. Never draw or subtract those on the same line.
4308
+ */
4309
+ benchmark_comparable: boolean;
4310
+ /**
4311
+ * Everything this version's run cost, in cents: training, the comparison
4312
+ * machine, the judge and the benchmark replay.
4313
+ */
4314
+ cost_cents: number;
4315
+ /**
4316
+ * True when no charge was recorded for the comparison machine (a run made
4317
+ * before the platform recorded one) and `cost_cents` counts it at its
4318
+ * ceiling instead, so the figure is the most the version can have cost,
4319
+ * not what it did.
4320
+ */
4321
+ cost_includes_candidate_ceiling: boolean;
4322
+ created_at: string;
4323
+ /**
4324
+ * When this version's head-to-head started. The champion it faced is the
4325
+ * model that was serving at that moment, which is why a delta against "the
4326
+ * champion before it" is decided by this and not by `created_at`. Null when
4327
+ * no comparison was ever started for it.
4328
+ */
4329
+ compared_at: string | null;
4330
+ /**
4331
+ * Why the run that made it fired. `variant` is a version trained on the same
4332
+ * conversations as the one before it with one setting changed; every other
4333
+ * value is a version that learned from newly reviewed conversations (or a
4334
+ * member's run now).
4335
+ */
4336
+ trigger: TrainingTrigger;
4337
+ /** See TrainingRun.recipe_note. */
4338
+ recipe_note: string | null;
4339
+ /** See TrainingRun.recipe_of_version. */
4340
+ recipe_of_version: number | null;
4341
+ /** See TrainingRun.recipe. */
4342
+ recipe: TrainingRecipe;
4343
+ finished_at: string | null;
4344
+ }
4345
+ /**
4346
+ * A training rule seen as the series of versions it produces: which exist,
4347
+ * which one serves, whether each beat the one before it, and what happens
4348
+ * next. READ-ONLY: every decision stays on the route that owns it (promote,
4349
+ * reject and roll back on the run; the version limit on the rule).
4350
+ */
4351
+ /**
4352
+ * Whether a pipeline can make a recipe variant next, and when it cannot, which
4353
+ * reason -- each is a different next step.
4354
+ *
4355
+ * - `off`: explore_recipes is off.
4356
+ * - `waiting_for_v1`: on, but a variant varies a version and there is none yet.
4357
+ * - `available`: on, with `recipe_variants_left` untried variants that fit.
4358
+ * - `no_slots`: untried variants fit (`recipe_variants_left` > 0), but every
4359
+ * version slot is made or taken by the run in flight. Raise `max_versions`
4360
+ * (or, at 10, start a new pipeline) and one can run. Only when variants are
4361
+ * left: a service that reports `no_slots` with `recipe_variants_left` of 0
4362
+ * means `all_tried`, and raising the limit starts nothing.
4363
+ * - `all_tried`: every variant that fits has been tried on the conversations
4364
+ * it has now, whether or not a version slot is free. Raising `max_versions`
4365
+ * does not start one; new reviews bring the next version.
4366
+ * - `none_fit`: no variant fits this rule's training settings.
4367
+ */
4368
+ export type PipelineRecipeExploration = 'off' | 'waiting_for_v1' | 'available' | 'no_slots' | 'all_tried' | 'none_fit';
4369
+ export interface Pipeline {
4370
+ rule_id: string;
4371
+ rule_name: string;
4372
+ enabled: boolean;
4373
+ /** The limit the member set, 1 to 10. */
4374
+ max_versions: number;
4375
+ /** Versions made so far. At `max_versions` the pipeline stops. */
4376
+ versions_made: number;
4377
+ /** The pinned base every version is a fresh adapter over, which is what makes their scores comparable. */
4378
+ base_model_id: string;
4379
+ base_model_revision: string;
4380
+ train_type: string;
4381
+ /** Null when what serves is not a version of this pipeline. */
4382
+ champion_version: number | null;
4383
+ /** The name the app calls. It does not change when a version is promoted. */
4384
+ serving_name: string;
4385
+ /**
4386
+ * Another pipeline that has since promoted onto the same serving name. Only
4387
+ * the most recent promotion serves; this pipeline's own champion no longer
4388
+ * does. Null when nothing has taken the name over.
4389
+ */
4390
+ serving_taken_over_by: string | null;
3984
4391
  benchmark_id: string | null;
4392
+ benchmark_name: string | null;
4393
+ benchmark_decides: boolean;
4394
+ benchmark_min_delta: number;
4395
+ status: PipelineStatus;
4396
+ /**
4397
+ * The rule's own sentence about what it is waiting for or why it stopped,
4398
+ * verbatim. Only as fresh as the platform's last visit: when `status` is
4399
+ * `paused`, `paused_reason` and `needs_consent` are what say why.
4400
+ */
4401
+ next_reason: string | null;
4402
+ last_checked_at: string | null;
4403
+ next_due_at: string | null;
4404
+ auto_promote: boolean;
4405
+ /** Why the platform stopped firing this rule, or null. Same codes as the rule's. */
4406
+ paused_reason: TrainingRulePausedReason | null;
4407
+ /**
4408
+ * The rule's terms changed and nobody has accepted them. It makes nothing
4409
+ * until somebody does.
4410
+ */
4411
+ needs_consent: boolean;
4412
+ /** The rule's explore_recipes, as it stands. */
4413
+ explore_recipes: boolean;
4414
+ /**
4415
+ * How many recipe variants that fit this rule's settings have not yet been
4416
+ * tried on the conversations the pipeline has now. Not capped by the free
4417
+ * version slots; 0 while exploration is off or when no variant fits. A count
4418
+ * is not a reason: read `recipe_exploration` for whether one can run next.
4419
+ */
4420
+ recipe_variants_left: number;
4421
+ /** Whether the next recipe variant can be tried, and if not, why. */
4422
+ recipe_exploration: PipelineRecipeExploration;
4423
+ /** The rule's monthly limit. Null means it has none. */
4424
+ monthly_ceiling_cents: number | null;
4425
+ /**
4426
+ * What this rule's runs have cost at most, counted toward the monthly limit
4427
+ * this UTC calendar month -- the figure the limit is enforced against. An
4428
+ * upper bound, not an exact spend. A run counts toward the month it was
4429
+ * created in, and a run still in progress also counts toward the current
4430
+ * month, at its three ceilings or what it has been billed when that is
4431
+ * more. A finished run counts what it was billed, and its comparison
4432
+ * machine is billed at the most it could have cost (the hourly cap for the
4433
+ * time it was up, never more than the candidate ceiling). A run created in
4434
+ * an earlier month that finishes in this one counts here only while it is
4435
+ * still running.
4436
+ */
4437
+ month_spent_cents: number;
4438
+ active_run: PipelineActiveRun | null;
4439
+ versions: PipelineVersion[];
3985
4440
  }
3986
4441
  export interface TrainingRuleListResponse {
3987
4442
  rules: TrainingRule[];
@@ -3990,6 +4445,7 @@ export interface TrainingRuleListResponse {
3990
4445
  export interface TrainingRuleResponse {
3991
4446
  rule: TrainingRule;
3992
4447
  recent_runs: TrainingRunSummary[];
4448
+ /** What this rule's runs have cost at most this UTC calendar month; see {@link Pipeline.month_spent_cents}. */
3993
4449
  month_spent_cents: number;
3994
4450
  judge_agreement: JudgeAgreement | null;
3995
4451
  }
@@ -4062,3 +4518,14 @@ export interface BenchmarkHistoryResponse {
4062
4518
  points: BenchmarkHistoryPoint[];
4063
4519
  total: number;
4064
4520
  }
4521
+ export interface PipelineListResponse {
4522
+ /** The newest pipelines first, at most the service's page (100). */
4523
+ pipelines: Pipeline[];
4524
+ /** Every pipeline in the workspace, including any not returned. */
4525
+ total: number;
4526
+ /** More pipelines exist than were returned; the rest are reached through `listTrainingRules`. */
4527
+ truncated: boolean;
4528
+ }
4529
+ export interface PipelineResponse {
4530
+ pipeline: Pipeline;
4531
+ }
package/dist/types.js CHANGED
@@ -12,3 +12,6 @@ export const TERMINAL_RUN_STATES = [
12
12
  'superseded',
13
13
  'rolled_back',
14
14
  ];
15
+ /* ── Pipelines ────────────────────────────────────────────────────────────── */
16
+ /** The most versions a pipeline may be set to make (loop-service maxVersionsCeiling). */
17
+ export const PIPELINE_MAX_VERSIONS_CEILING = 10;
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "runbios-sdk",
3
- "version": "0.2.13-dev.242",
3
+ "version": "0.2.13-dev.243",
4
4
  "description": "Official TypeScript SDK for the Run BiOS training and deployment platform API",
5
5
  "type": "module",
6
6
  "main": "./dist/index.js",