runbios-sdk 0.2.1-rc.151 → 0.2.1-rc.156

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/dist/index.d.ts CHANGED
@@ -36,7 +36,7 @@ import { Loop } from './resources/loop.js';
36
36
  * SDK version. Sent as part of the User-Agent header.
37
37
  * Must match package.json "version" -- enforced by a contract test.
38
38
  */
39
- export declare const VERSION = "0.2.1-rc.151";
39
+ export declare const VERSION = "0.2.1-rc.156";
40
40
  export declare class RunBiOS {
41
41
  /** Search models, fetch configs, check adapter compatibility. */
42
42
  readonly models: Models;
@@ -75,6 +75,6 @@ export { Training } from './resources/training.js';
75
75
  export { Wallet } from './resources/wallet.js';
76
76
  export { GPU, type GPURecommendation } from './resources/gpu.js';
77
77
  export { Inference, validateChatRequest, parseSSE, buildInferenceRequest, contextSizingBasis, type ContextSizingBasis, CONTEXT_DEFAULT_CEILING, CONTEXT_EDITABLE_FLOOR, type ChatMessage, type FunctionTool, type ChatCompletionParams, type ChatCompletionResponse, type ChatCompletionChunk, } from './resources/inference.js';
78
- export type { BiOSConfig, PaginatedResponse, ApiErrorBody, AvailableGpuAlternative, CapacityMinimumRequirement, InferenceBookingAccepted, Model, ModelSearchParams, ModelSearchResponse, ModelDetailResponse, ModelConfig, Dataset, DatasetListParams, DatasetListResponse, DatasetUploadParams, DatasetPreview, DatasetPreviewParams, DatasetImportHFParams, DatasetRegisterHFParams, DatasetHubSearchParams, DatasetHubPreviewParams, DatasetValidation, DatasetFormatVariant, DatasetFormatSpec, DatasetFormatSpecs, DatasetStorageUsage, TrainingMethod, RLHFAlgorithm, AdapterType, TrainingJobStatus, TrainingStatusFilter, TrainingCreateParams, TrainingListParams, TrainingListResponse, TrainingJob, TrainingMetrics, TrainingResourceMetrics, TrainingDeviceMetrics, MetricPoint, MetricGraphConfig, TrainingCheckpoint, TrainingLogs, TrainingLogEntry, TrainingStopResponse, TrainingResumeResponse, CanonicalTrainingRequest, TrainingPreflightDataset, TrainingPreflightWarning, TrainingPreflightResponse, TrainingCapabilityChoice, TrainingConfigFieldCapability, TrainingCapabilities, GPUChoice, WalletBalance, Transaction, TransactionListResponse, TransactionListParams, GPUInfo, GPUPricingResponse, GPUOptionsParams, TrainingRecommendParams, TrainingAdvisorVerdict, TrainingAdvisorRecommendation, TrainingAdvisorUserShape, TrainingAdvisorJustification, TrainingAdvisorBasis, GPUOption, GPUOptionSuggestion, GPUOptionsResponse, InferenceStatus, InferenceCreateParams, InferenceUpdateParams, InferenceDeployment, InferenceDeploymentSummary, InferenceCreateResponse, InferenceListResponse, InferenceUpdateResponse, InferenceLifecycleResponse, InferenceDeleteResponse, InferenceGPUOptionMarket, InferenceGPUOption, InferenceGPUAlternative, InferenceGPUOptionsParams, InferenceGPUOptionsResponse, InferenceCanonicalRequest, InferencePreflightWarning, InferencePreflightResponse, AdapterCompatibility, AdapterCompatibilityResponse, AdapterCompatibilityParams, SupportedArchitecture, ArchitectureScope, SupportedArchitecturesResponse, ApiKey, ApiKeyScope, ApiKeyIntrospection, Organization, OrgMember, OrgInvite, Workspace, Integration, IntegrationCreateParams, StorageUsage, StorageObject, PromptTokensDetails, ChatCompletionUsage, TrainingCadence, TrainingCombinator, ServingKind, TrainingRulePausedReason, LoopTrainingMethod, TrainType, GradersScope, TrainingTrigger, TrainingRunState, TrainingVerdict, TrainingDecision, NotifyState, EvaluationStatus, EvaluationItemStatus, EvaluationWinner, EvaluationWarningCode, ConsentVia, TrainingToolCall, TrainingMessage, TrainingRuleServing, TrainingGPURung, TrainingRule, TrainingRuleConsent, TrainingRulePreflightRefusal, TrainingRuleKeyRef, TrainingRuleKeyCheck, TrainingRulePreflight, TrainingRuleBuildSpec, TrainingRuleTriggerInput, TrainingRuleTrainingInput, TrainingRuleDeployInput, TrainingRuleMoneyInput, TrainingRuleEvaluationInput, TrainingRulePromotionInput, TrainingRuleAcceptTerms, TrainingRulePreflightRequest, TrainingRuleCreateRequest, TrainingRuleUpdateRequest, TrainingRuleConsentRequest, TrainingRuleListParams, TrainingRunPromoteRequest, TrainingRunRejectRequest, TrainingRunRollbackRequest, TrainingRunCancelRequest, TrainingRunListParams, TrainingRun, TrainingRunSummary, TrainingRunEvent, TrainingRunLinks, TrainingRunActions, EvaluationRef, EvaluationDecoding, EvaluationJudgeDimension, EvaluationDimensionScore, EvaluationGraderScore, EvaluationWarning, EvaluationMargin, JudgeAgreementDimension, JudgeAgreement, Evaluation, EvaluationItemSide, EvaluationItem, EvaluationItemListParams, JudgeAgreementParams, AgentSettings, AgentSettingsRequest, InferenceAlias, InferenceAliasRequest, TrainingRuleListResponse, TrainingRuleResponse, TrainingRuleMutationResponse, TrainingRuleDeleteResponse, TrainingRunListResponse, TrainingRunResponse, TrainingRunActionResponse, EvaluationItemsResponse, AgentSettingsResponse, InferenceAliasListResponse, InferenceAliasResponse, InferenceAliasDeleteResponse, } from './types.js';
78
+ export type { BiOSConfig, PaginatedResponse, ApiErrorBody, AvailableGpuAlternative, CapacityMinimumRequirement, InferenceBookingAccepted, Model, ModelSearchParams, ModelSearchResponse, ModelDetailResponse, ModelConfig, Dataset, DatasetListParams, DatasetListResponse, DatasetUploadParams, DatasetPreview, DatasetPreviewParams, DatasetImportHFParams, DatasetRegisterHFParams, DatasetHubSearchParams, DatasetHubPreviewParams, DatasetValidation, DatasetFormatVariant, DatasetFormatSpec, DatasetFormatSpecs, DatasetStorageUsage, TrainingMethod, RLHFAlgorithm, AdapterType, TrainingJobStatus, TrainingStatusFilter, TrainingCreateParams, TrainingListParams, TrainingListResponse, TrainingJob, TrainingMetrics, TrainingResourceMetrics, TrainingDeviceMetrics, MetricPoint, MetricGraphConfig, TrainingCheckpoint, TrainingLogs, TrainingLogEntry, TrainingStopResponse, TrainingResumeResponse, CanonicalTrainingRequest, TrainingPreflightDataset, TrainingPreflightWarning, TrainingPreflightResponse, TrainingCapabilityChoice, TrainingConfigFieldCapability, TrainingCapabilities, GPUChoice, WalletBalance, Transaction, TransactionListResponse, TransactionListParams, GPUInfo, GPUPricingResponse, GPUOptionsParams, TrainingRecommendParams, TrainingAdvisorVerdict, TrainingAdvisorRecommendation, TrainingAdvisorUserShape, TrainingAdvisorJustification, TrainingAdvisorBasis, GPUOption, GPUOptionSuggestion, GPUOptionsResponse, InferenceStatus, InferenceCreateParams, InferenceUpdateParams, InferenceDeployment, InferenceDeploymentSummary, InferenceCreateResponse, InferenceListResponse, InferenceUpdateResponse, InferenceLifecycleResponse, InferenceDeleteResponse, InferenceGPUOptionMarket, InferenceGPUOption, InferenceGPUAlternative, InferenceGPUOptionsParams, InferenceGPUOptionsResponse, InferenceCanonicalRequest, InferencePreflightWarning, InferencePreflightResponse, AdapterCompatibility, AdapterCompatibilityResponse, AdapterCompatibilityParams, SupportedArchitecture, ArchitectureScope, SupportedArchitecturesResponse, ApiKey, ApiKeyScope, ApiKeyIntrospection, Organization, OrgMember, OrgInvite, Workspace, Integration, IntegrationCreateParams, StorageUsage, StorageObject, PromptTokensDetails, ChatCompletionUsage, TrainingCadence, TrainingCombinator, ServingKind, TrainingRulePausedReason, LoopTrainingMethod, TrainType, GradersScope, TrainingTrigger, TrainingRunState, TrainingVerdict, TrainingDecision, NotifyState, EvaluationStatus, EvaluationItemStatus, EvaluationWinner, EvaluationWarningCode, ConsentVia, TrainingToolCall, TrainingMessage, TrainingRuleServing, TrainingGPURung, TrainingRule, TrainingRuleConsent, TrainingRulePreflightRefusal, TrainingRuleKeyRef, TrainingRuleKeyCheck, TrainingRulePreflight, TrainingRuleBuildSpec, TrainingRuleTriggerInput, TrainingRuleTrainingInput, TrainingRuleDeployInput, TrainingRuleMoneyInput, TrainingRuleEvaluationInput, TrainingRulePromotionInput, TrainingRuleAcceptTerms, TrainingRulePreflightRequest, TrainingRuleCreateRequest, TrainingRuleUpdateRequest, TrainingRuleConsentRequest, TrainingRuleListParams, TrainingRunPromoteRequest, TrainingRunRejectRequest, TrainingRunRollbackRequest, TrainingRunCancelRequest, TrainingRunListParams, TrainingRun, TrainingRunSummary, TrainingRunEvent, TrainingRunLinks, TrainingRunActions, EvaluationRef, EvaluationDecoding, EvaluationJudgeDimension, EvaluationDimensionScore, EvaluationGraderScore, EvaluationWarning, EvaluationMargin, JudgeAgreementDimension, JudgeAgreement, Evaluation, EvaluationItemSide, EvaluationItem, EvaluationItemListParams, JudgeAgreementParams, AgentSettings, AgentSettingsRequest, BenchmarkStatus, BenchmarkSourceKind, BenchmarkRunStatus, BenchmarkScoring, BenchmarkDimensionScore, Benchmark, BenchmarkItem, BenchmarkRun, BenchmarkHistoryPoint, BenchmarkSource, BenchmarkCreateParams, BenchmarkListParams, BenchmarkItemListParams, BenchmarkHistoryParams, BenchmarkRetireParams, TrainingRuleBenchmarkRequest, InferenceAlias, InferenceAliasRequest, TrainingRuleListResponse, TrainingRuleResponse, TrainingRuleMutationResponse, TrainingRuleDeleteResponse, TrainingRunListResponse, TrainingRunResponse, TrainingRunActionResponse, EvaluationItemsResponse, AgentSettingsResponse, InferenceAliasListResponse, InferenceAliasResponse, InferenceAliasDeleteResponse, BenchmarkListResponse, BenchmarkItemsResponse, BenchmarkHistoryResponse, } from './types.js';
79
79
  /** The closed set of run states a training run never leaves. */
80
80
  export { TERMINAL_RUN_STATES } from './types.js';
package/dist/index.js CHANGED
@@ -36,7 +36,7 @@ import { Loop } from './resources/loop.js';
36
36
  * SDK version. Sent as part of the User-Agent header.
37
37
  * Must match package.json "version" -- enforced by a contract test.
38
38
  */
39
- export const VERSION = '0.2.1-rc.151';
39
+ export const VERSION = '0.2.1-rc.156';
40
40
  export class RunBiOS {
41
41
  /** Search models, fetch configs, check adapter compatibility. */
42
42
  models;
@@ -1,5 +1,5 @@
1
1
  import type { HttpClient } from '../client.js';
2
- import type { LoopTrace, LoopTraceListResponse, LoopTraceListParams, LoopCaptureParams, LoopCaptureResult, LoopImportParams, LoopImportResult, LoopSignal, LoopSignalParams, LoopCandidate, LoopCandidateParams, LoopGrader, LoopGraderParams, LoopLabel, LoopLabelCount, LoopGradeResult, LoopDataset, LoopDatasetListParams, LoopDatasetItem, LoopDatasetCreateParams, LoopConfig, LoopConfigParams, LoopBuildRule, LoopBuildRuleParams, LoopBuildRuleUpdateParams, LoopStats, LoopJudge, LoopJudgeParams, LoopAgentCredential, LoopAgentStatus, LoopSampleRun, LoopSampleRunParams, LoopJudgeRun, LoopJudgeRunItems, LoopJudgeRunItemStatus, LoopJudgeWork, LoopJudgeVerdict, LoopJudgeVerdictResult, TrainingRule, TrainingRulePreflight, TrainingRulePreflightRequest, TrainingRuleCreateRequest, TrainingRuleUpdateRequest, TrainingRuleConsentRequest, TrainingRuleListParams, TrainingRuleListResponse, TrainingRuleResponse, TrainingRuleMutationResponse, TrainingRuleDeleteResponse, TrainingRun, TrainingRunListParams, TrainingRunListResponse, TrainingRunResponse, TrainingRunActionResponse, TrainingRunPromoteRequest, TrainingRunRejectRequest, TrainingRunRollbackRequest, TrainingRunCancelRequest, Evaluation, EvaluationItemListParams, EvaluationItemsResponse, JudgeAgreement, JudgeAgreementParams, AgentSettings, AgentSettingsRequest } from '../types.js';
2
+ import type { LoopTrace, LoopTraceListResponse, LoopTraceListParams, LoopCaptureParams, LoopCaptureResult, LoopImportParams, LoopImportResult, LoopSignal, LoopSignalParams, LoopCandidate, LoopCandidateParams, LoopGrader, LoopGraderParams, LoopLabel, LoopLabelCount, LoopGradeResult, LoopDataset, LoopDatasetListParams, LoopDatasetItem, LoopDatasetCreateParams, LoopConfig, LoopConfigParams, LoopBuildRule, LoopBuildRuleParams, LoopBuildRuleUpdateParams, LoopStats, LoopJudge, LoopJudgeParams, LoopAgentCredential, LoopAgentStatus, LoopSampleRun, LoopSampleRunParams, LoopJudgeRun, LoopJudgeRunItems, LoopJudgeRunItemStatus, LoopJudgeWork, LoopJudgeVerdict, LoopJudgeVerdictResult, TrainingRule, TrainingRulePreflight, TrainingRulePreflightRequest, TrainingRuleCreateRequest, TrainingRuleUpdateRequest, TrainingRuleConsentRequest, TrainingRuleListParams, TrainingRuleListResponse, TrainingRuleResponse, TrainingRuleMutationResponse, TrainingRuleDeleteResponse, TrainingRun, TrainingRunListParams, TrainingRunListResponse, TrainingRunResponse, TrainingRunActionResponse, TrainingRunPromoteRequest, TrainingRunRejectRequest, TrainingRunRollbackRequest, TrainingRunCancelRequest, Evaluation, EvaluationItemListParams, EvaluationItemsResponse, JudgeAgreement, JudgeAgreementParams, Benchmark, BenchmarkCreateParams, BenchmarkHistoryParams, BenchmarkHistoryResponse, BenchmarkItemListParams, BenchmarkItemsResponse, BenchmarkListParams, BenchmarkListResponse, BenchmarkRetireParams, BenchmarkRun, TrainingRuleBenchmarkRequest, AgentSettings, AgentSettingsRequest } from '../types.js';
3
3
  /**
4
4
  * The Conscious Loop -- capture what your model was asked and answered, record
5
5
  * whether it was right, and turn those judgements into training data.
@@ -567,10 +567,11 @@ export declare class Loop {
567
567
  * `valid: false` with an EMPTY `refusals` list is not nothing: check
568
568
  * `unreachable`, which names every peer the platform could not reach. The
569
569
  * rule is savable in that state and would be paused, but the estimate around
570
- * it is not trustworthy -- both `worst_hourly_*` are `0` -- so do not quote
571
- * those figures to anyone. `model_revision` is NOT a signal here: a degraded
572
- * pass echoes back whatever revision the request carried, so test
573
- * `unreachable.length > 0` rather than the revision being empty.
570
+ * it is not trustworthy -- both `worst_hourly_*` are `0`, and so are both
571
+ * `max_*_hours` beside them -- so do not quote those figures to anyone.
572
+ * `model_revision` is empty unless training-service actually pinned a
573
+ * commit; test `unreachable.length > 0` to tell a degraded pass from a
574
+ * refusal.
574
575
  */
575
576
  preflightTrainingRule(params: TrainingRulePreflightRequest): Promise<TrainingRulePreflight>;
576
577
  /**
@@ -739,6 +740,98 @@ export declare class Loop {
739
740
  * reviewer overlap yet" is an answer, and 100% of two pairs is not.
740
741
  */
741
742
  getJudgeAgreement(judgeId: string, params?: JudgeAgreementParams): Promise<JudgeAgreement>;
743
+ /**
744
+ * Pin a fixed set of conversations, with a judge frozen beside them, and
745
+ * measure every future run against it.
746
+ *
747
+ * The comparison answers "is this candidate better than what serves today".
748
+ * It cannot answer "is my model getting better", because its rows, its
749
+ * opponent and its judge all move between runs. A benchmark is the other
750
+ * instrument: the same conversations, the same oracle, the same decoding,
751
+ * replayed against both models on every run that finishes its comparison,
752
+ * and reported as two absolute numbers on a scale that does not move.
753
+ *
754
+ * 10 to 200 conversations. A named conversation with nothing to ask a model
755
+ * is refused rather than skipped, because pinning 47 of the 50 you chose is
756
+ * the set being wrong from the first day and you would never find out.
757
+ *
758
+ * It raises no amount you have already agreed to. The replay's calls come
759
+ * out of the rule's existing `eval_ceiling_cents`, and
760
+ * `per_run_ceiling_cents` can only lower what is spent inside that.
761
+ */
762
+ createBenchmark(params: BenchmarkCreateParams): Promise<Benchmark>;
763
+ /** Your benchmarks, newest first. Filter by `status` to hide retired ones. */
764
+ listBenchmarks(params?: BenchmarkListParams): Promise<BenchmarkListResponse>;
765
+ /**
766
+ * One benchmark: the frozen oracle, the scoring rule and the fingerprint of
767
+ * the set.
768
+ *
769
+ * `items_digest` is recomputed from the rows and compared at the start of
770
+ * every replay, so "the set cannot drift" is something you can check rather
771
+ * than something the platform promises.
772
+ */
773
+ getBenchmark(id: string): Promise<Benchmark>;
774
+ /**
775
+ * The pinned conversations, paged. `limit` is capped at 100 by the service.
776
+ *
777
+ * `source_trace_id` and `source_trace_url` come back null once the
778
+ * conversation a row was copied from has been deleted. The row itself stays
779
+ * and every number already measured against it stays exactly as comparable
780
+ * as it was: retention cannot shrink a benchmark.
781
+ */
782
+ listBenchmarkItems(id: string, params?: BenchmarkItemListParams): Promise<BenchmarkItemsResponse>;
783
+ /**
784
+ * The trend: every score this benchmark has produced, newest first, each
785
+ * beside the checkpoint that produced it and what you then did about it.
786
+ *
787
+ * EVERY REPLAY IS A POINT, including the ones with no numbers on them. A
788
+ * series that silently dropped the runs where the benchmark could not be
789
+ * scored would read as an unbroken line and not be one, so those points come
790
+ * back with `candidate_score` null and `status_reason` saying why.
791
+ */
792
+ getBenchmarkHistory(id: string, params?: BenchmarkHistoryParams): Promise<BenchmarkHistoryResponse>;
793
+ /**
794
+ * Stop replaying a benchmark, and detach it from every rule that names it.
795
+ *
796
+ * Nothing measured is removed. There is no delete and no update on purpose:
797
+ * the series of numbers measured against a set is what a benchmark is for,
798
+ * so an edit would make everything before it incomparable with everything
799
+ * after it and a delete would throw the series away. Changing the set means
800
+ * pinning a new benchmark, and retiring the old one releases its name.
801
+ *
802
+ * Retiring twice is somebody pressing a button twice: the second call
803
+ * answers with the retired benchmark rather than a refusal.
804
+ */
805
+ retireBenchmark(id: string, params?: BenchmarkRetireParams): Promise<Benchmark>;
806
+ /**
807
+ * One replay in full: both absolute scores, the difference between them on
808
+ * the same set, the per-dimension breakdown and what it cost.
809
+ *
810
+ * Read `status` before you read the scores. A replay that did not score
811
+ * every pinned conversation publishes no score at all -- all three of
812
+ * `candidate_score`, `incumbent_score` and `score_delta` are null and
813
+ * `status_reason` is the sentence that explains it -- because a mean over
814
+ * whichever conversations happened to succeed is a measurement of a
815
+ * different set, which is the exact defect a standing benchmark removes.
816
+ */
817
+ getBenchmarkRun(id: string): Promise<BenchmarkRun>;
818
+ /**
819
+ * Attach a benchmark to a rule, or send `{ benchmark_id: null }` to detach
820
+ * it.
821
+ *
822
+ * Its own route rather than a field on the rule body, because it is a
823
+ * decision to replay a fixed set on every future run of this rule for as
824
+ * long as it stands, and its refusals -- retired, or belonging to another
825
+ * workspace -- are about the benchmark rather than about the rule.
826
+ *
827
+ * It does not invalidate consent and the reply says so: attaching raises
828
+ * neither the amount set aside for judge calls nor the amount set aside for
829
+ * keeping the new model available, so nobody is asked to read the same
830
+ * sentence again. A retired benchmark is refused with `BENCHMARK_RETIRED`,
831
+ * because a rule pointed at one would report no number on every run and say
832
+ * nothing about why.
833
+ */
834
+ setTrainingRuleBenchmark(id: string, params: TrainingRuleBenchmarkRequest): Promise<TrainingRuleMutationResponse>;
742
835
  /**
743
836
  * The workspace's agent options: which model it defaults to, the system
744
837
  * prompts it judges and samples with, and the monthly cap on what its model
@@ -776,10 +776,11 @@ export class Loop {
776
776
  * `valid: false` with an EMPTY `refusals` list is not nothing: check
777
777
  * `unreachable`, which names every peer the platform could not reach. The
778
778
  * rule is savable in that state and would be paused, but the estimate around
779
- * it is not trustworthy -- both `worst_hourly_*` are `0` -- so do not quote
780
- * those figures to anyone. `model_revision` is NOT a signal here: a degraded
781
- * pass echoes back whatever revision the request carried, so test
782
- * `unreachable.length > 0` rather than the revision being empty.
779
+ * it is not trustworthy -- both `worst_hourly_*` are `0`, and so are both
780
+ * `max_*_hours` beside them -- so do not quote those figures to anyone.
781
+ * `model_revision` is empty unless training-service actually pinned a
782
+ * commit; test `unreachable.length > 0` to tell a degraded pass from a
783
+ * refusal.
783
784
  */
784
785
  async preflightTrainingRule(params) {
785
786
  return this._http.fetchPost('/api/loop/training-rules/preflight', params);
@@ -1030,6 +1031,139 @@ export class Loop {
1030
1031
  const qs = q.toString();
1031
1032
  return this._http.fetchGet(`/api/loop/judges/${encodeURIComponent(judgeId)}/agreement${qs ? `?${qs}` : ''}`);
1032
1033
  }
1034
+ // ── the standing benchmark ────────────────────────────────────────────
1035
+ /**
1036
+ * Pin a fixed set of conversations, with a judge frozen beside them, and
1037
+ * measure every future run against it.
1038
+ *
1039
+ * The comparison answers "is this candidate better than what serves today".
1040
+ * It cannot answer "is my model getting better", because its rows, its
1041
+ * opponent and its judge all move between runs. A benchmark is the other
1042
+ * instrument: the same conversations, the same oracle, the same decoding,
1043
+ * replayed against both models on every run that finishes its comparison,
1044
+ * and reported as two absolute numbers on a scale that does not move.
1045
+ *
1046
+ * 10 to 200 conversations. A named conversation with nothing to ask a model
1047
+ * is refused rather than skipped, because pinning 47 of the 50 you chose is
1048
+ * the set being wrong from the first day and you would never find out.
1049
+ *
1050
+ * It raises no amount you have already agreed to. The replay's calls come
1051
+ * out of the rule's existing `eval_ceiling_cents`, and
1052
+ * `per_run_ceiling_cents` can only lower what is spent inside that.
1053
+ */
1054
+ async createBenchmark(params) {
1055
+ const res = await this._http.fetchPost('/api/loop/benchmarks', params);
1056
+ return res.benchmark;
1057
+ }
1058
+ /** Your benchmarks, newest first. Filter by `status` to hide retired ones. */
1059
+ async listBenchmarks(params = {}) {
1060
+ const q = new URLSearchParams();
1061
+ if (params.status)
1062
+ q.set('status', params.status);
1063
+ if (params.limit != null)
1064
+ q.set('limit', String(params.limit));
1065
+ if (params.offset != null)
1066
+ q.set('offset', String(params.offset));
1067
+ const qs = q.toString();
1068
+ return this._http.fetchGet(`/api/loop/benchmarks${qs ? `?${qs}` : ''}`);
1069
+ }
1070
+ /**
1071
+ * One benchmark: the frozen oracle, the scoring rule and the fingerprint of
1072
+ * the set.
1073
+ *
1074
+ * `items_digest` is recomputed from the rows and compared at the start of
1075
+ * every replay, so "the set cannot drift" is something you can check rather
1076
+ * than something the platform promises.
1077
+ */
1078
+ async getBenchmark(id) {
1079
+ const res = await this._http.fetchGet(`/api/loop/benchmarks/${encodeURIComponent(id)}`);
1080
+ return res.benchmark;
1081
+ }
1082
+ /**
1083
+ * The pinned conversations, paged. `limit` is capped at 100 by the service.
1084
+ *
1085
+ * `source_trace_id` and `source_trace_url` come back null once the
1086
+ * conversation a row was copied from has been deleted. The row itself stays
1087
+ * and every number already measured against it stays exactly as comparable
1088
+ * as it was: retention cannot shrink a benchmark.
1089
+ */
1090
+ async listBenchmarkItems(id, params = {}) {
1091
+ const q = new URLSearchParams();
1092
+ if (params.limit != null)
1093
+ q.set('limit', String(params.limit));
1094
+ if (params.offset != null)
1095
+ q.set('offset', String(params.offset));
1096
+ const qs = q.toString();
1097
+ return this._http.fetchGet(`/api/loop/benchmarks/${encodeURIComponent(id)}/items${qs ? `?${qs}` : ''}`);
1098
+ }
1099
+ /**
1100
+ * The trend: every score this benchmark has produced, newest first, each
1101
+ * beside the checkpoint that produced it and what you then did about it.
1102
+ *
1103
+ * EVERY REPLAY IS A POINT, including the ones with no numbers on them. A
1104
+ * series that silently dropped the runs where the benchmark could not be
1105
+ * scored would read as an unbroken line and not be one, so those points come
1106
+ * back with `candidate_score` null and `status_reason` saying why.
1107
+ */
1108
+ async getBenchmarkHistory(id, params = {}) {
1109
+ const q = new URLSearchParams();
1110
+ if (params.limit != null)
1111
+ q.set('limit', String(params.limit));
1112
+ if (params.offset != null)
1113
+ q.set('offset', String(params.offset));
1114
+ const qs = q.toString();
1115
+ return this._http.fetchGet(`/api/loop/benchmarks/${encodeURIComponent(id)}/history${qs ? `?${qs}` : ''}`);
1116
+ }
1117
+ /**
1118
+ * Stop replaying a benchmark, and detach it from every rule that names it.
1119
+ *
1120
+ * Nothing measured is removed. There is no delete and no update on purpose:
1121
+ * the series of numbers measured against a set is what a benchmark is for,
1122
+ * so an edit would make everything before it incomparable with everything
1123
+ * after it and a delete would throw the series away. Changing the set means
1124
+ * pinning a new benchmark, and retiring the old one releases its name.
1125
+ *
1126
+ * Retiring twice is somebody pressing a button twice: the second call
1127
+ * answers with the retired benchmark rather than a refusal.
1128
+ */
1129
+ async retireBenchmark(id, params = {}) {
1130
+ const res = await this._http.fetchPost(`/api/loop/benchmarks/${encodeURIComponent(id)}/retire`, params);
1131
+ return res.benchmark;
1132
+ }
1133
+ /**
1134
+ * One replay in full: both absolute scores, the difference between them on
1135
+ * the same set, the per-dimension breakdown and what it cost.
1136
+ *
1137
+ * Read `status` before you read the scores. A replay that did not score
1138
+ * every pinned conversation publishes no score at all -- all three of
1139
+ * `candidate_score`, `incumbent_score` and `score_delta` are null and
1140
+ * `status_reason` is the sentence that explains it -- because a mean over
1141
+ * whichever conversations happened to succeed is a measurement of a
1142
+ * different set, which is the exact defect a standing benchmark removes.
1143
+ */
1144
+ async getBenchmarkRun(id) {
1145
+ const res = await this._http.fetchGet(`/api/loop/benchmark-runs/${encodeURIComponent(id)}`);
1146
+ return res.benchmark_run;
1147
+ }
1148
+ /**
1149
+ * Attach a benchmark to a rule, or send `{ benchmark_id: null }` to detach
1150
+ * it.
1151
+ *
1152
+ * Its own route rather than a field on the rule body, because it is a
1153
+ * decision to replay a fixed set on every future run of this rule for as
1154
+ * long as it stands, and its refusals -- retired, or belonging to another
1155
+ * workspace -- are about the benchmark rather than about the rule.
1156
+ *
1157
+ * It does not invalidate consent and the reply says so: attaching raises
1158
+ * neither the amount set aside for judge calls nor the amount set aside for
1159
+ * keeping the new model available, so nobody is asked to read the same
1160
+ * sentence again. A retired benchmark is refused with `BENCHMARK_RETIRED`,
1161
+ * because a rule pointed at one would report no number on every run and say
1162
+ * nothing about why.
1163
+ */
1164
+ async setTrainingRuleBenchmark(id, params) {
1165
+ return this._http.fetchPut(`/api/loop/training-rules/${encodeURIComponent(id)}/benchmark`, params);
1166
+ }
1033
1167
  // ── agent settings ────────────────────────────────────────────────────
1034
1168
  /**
1035
1169
  * The workspace's agent options: which model it defaults to, the system
package/dist/types.d.ts CHANGED
@@ -3083,6 +3083,15 @@ export interface TrainingRule {
3083
3083
  eval_max_rows: number;
3084
3084
  eval_max_tokens: number;
3085
3085
  min_holdout_rows: number;
3086
+ /**
3087
+ * The standing benchmark replayed on every run of this rule, beside the
3088
+ * per-run comparison and never instead of it. Null is the ordinary state.
3089
+ *
3090
+ * Set with `setTrainingRuleBenchmark`, not by an update to the rule: it is a
3091
+ * decision to replay a fixed set on every future run, and it has refusals of
3092
+ * its own. It raises no amount, so it does not invalidate consent.
3093
+ */
3094
+ benchmark_id: string | null;
3086
3095
  auto_promote: boolean;
3087
3096
  promote_margin: number;
3088
3097
  promote_min_win_rate: number;
@@ -3352,6 +3361,15 @@ export interface TrainingRun {
3352
3361
  candidate_deadline_at: string | null;
3353
3362
  candidate_cleanup_at: string | null;
3354
3363
  evaluation_id: string | null;
3364
+ /** This run's replay of the rule's standing benchmark, if it had one. */
3365
+ benchmark_run_id: string | null;
3366
+ /**
3367
+ * What the replay's calls cost. Kept apart from `spent_eval_cents` so a
3368
+ * report can say what the comparison cost and what the benchmark cost; both
3369
+ * come out of `eval_ceiling_cents` and their sum can never exceed it, so
3370
+ * anything totalling what a run cost has to add this one too.
3371
+ */
3372
+ benchmark_spent_cents: number;
3355
3373
  alias_name: string | null;
3356
3374
  alias_written_at: string | null;
3357
3375
  serving_before: TrainingRuleServing | null;
@@ -3389,6 +3407,8 @@ export interface TrainingRunSummary {
3389
3407
  billed_training_cents: number;
3390
3408
  billed_candidate_cents: number;
3391
3409
  spent_eval_cents: number;
3410
+ /** The fourth money column. A row that leaves it out adds up short. */
3411
+ benchmark_spent_cents: number;
3392
3412
  training_job_id: string | null;
3393
3413
  candidate_deployment_id: string | null;
3394
3414
  evaluation_id: string | null;
@@ -3414,6 +3434,8 @@ export interface TrainingRunLinks {
3414
3434
  candidate_url: string | null;
3415
3435
  dataset_url: string | null;
3416
3436
  evaluation_url: string | null;
3437
+ /** The TREND the benchmark number belongs to, not the one replay. */
3438
+ benchmark_url: string | null;
3417
3439
  }
3418
3440
  /** What this reader may do right now. A button that cannot work is never shown. */
3419
3441
  export interface TrainingRunActions {
@@ -3614,6 +3636,203 @@ export interface InferenceAliasRequest {
3614
3636
  target_inference_id: string;
3615
3637
  origin?: string;
3616
3638
  }
3639
+ /**
3640
+ * A benchmark is active until it is retired. There is no delete and no update:
3641
+ * the series of numbers measured against a set is what a benchmark is for, so
3642
+ * an edit would make every number before it incomparable with every number
3643
+ * after it, and a delete throws the series away.
3644
+ */
3645
+ export type BenchmarkStatus = 'active' | 'retired';
3646
+ /** Where the pinned conversations are copied from. Read once, at creation. */
3647
+ export type BenchmarkSourceKind = 'traces' | 'dataset';
3648
+ /**
3649
+ * The life of one replay.
3650
+ *
3651
+ * `budget_stopped` reached the amount left for it inside the run's own ceiling,
3652
+ * and `abandoned` ran out of the time the run sets aside for it. Both are
3653
+ * points on the trend carrying `status_reason`, never silent gaps.
3654
+ */
3655
+ export type BenchmarkRunStatus = 'open' | 'done' | 'failed' | 'budget_stopped' | 'abandoned';
3656
+ /**
3657
+ * How the pinned conversations become one number, frozen on the benchmark.
3658
+ *
3659
+ * `require_all_rows` is what makes the number comparable at all: a mean over
3660
+ * whichever conversations happened to succeed is a measurement of a different
3661
+ * set, so a short replay publishes no score and says why. A partial benchmark
3662
+ * is a missing number, never a lower one.
3663
+ */
3664
+ export interface BenchmarkScoring {
3665
+ metric: string;
3666
+ scale: string;
3667
+ aggregate: string;
3668
+ require_all_rows: boolean;
3669
+ }
3670
+ /**
3671
+ * One rubric dimension of one replay, for both models.
3672
+ *
3673
+ * Deliberately not `EvaluationDimensionScore`: these are absolute means on a
3674
+ * fixed set and those are paired means on that run's own held-back rows.
3675
+ */
3676
+ export interface BenchmarkDimensionScore {
3677
+ dimension: string;
3678
+ incumbent: number;
3679
+ candidate: number;
3680
+ delta: number;
3681
+ }
3682
+ /**
3683
+ * The frozen definition: the oracle, the scoring rule and the fingerprint of
3684
+ * the set. The conversations themselves are their own paged route.
3685
+ */
3686
+ export interface Benchmark {
3687
+ id: string;
3688
+ workspace_id: string;
3689
+ name: string;
3690
+ status: BenchmarkStatus;
3691
+ /** Provenance only: everything needed from the judge is copied below it. */
3692
+ judge_id: string | null;
3693
+ judge_name: string;
3694
+ judge_model: string;
3695
+ judge_instructions: string;
3696
+ judge_dimensions: EvaluationJudgeDimension[];
3697
+ judge_system_prompt: string | null;
3698
+ decoding: EvaluationDecoding;
3699
+ scoring: BenchmarkScoring;
3700
+ item_count: number;
3701
+ /** Recomputed from the rows and compared at the start of every replay. */
3702
+ items_digest: string;
3703
+ /** A cap INSIDE the run's own amount for judge calls, never an addition to it. */
3704
+ per_run_ceiling_cents: number;
3705
+ created_by: string;
3706
+ created_at: string;
3707
+ retired_at: string | null;
3708
+ retired_by: string | null;
3709
+ }
3710
+ /**
3711
+ * One pinned conversation.
3712
+ *
3713
+ * `source_trace_id` is a link and is allowed to go null: the prompt was copied
3714
+ * at pin time, so retention removing the conversation takes away the ability to
3715
+ * open it and takes away nothing else. Every number already measured stays
3716
+ * exactly as comparable as it was.
3717
+ */
3718
+ export interface BenchmarkItem {
3719
+ id: number;
3720
+ ordinal: number;
3721
+ source_trace_id: string | null;
3722
+ source_trace_url: string | null;
3723
+ prompt: TrainingMessage[];
3724
+ tools: unknown[] | null;
3725
+ }
3726
+ /**
3727
+ * One replay: this benchmark, on this training run, against both models.
3728
+ *
3729
+ * IT DOES NOT GATE PROMOTION. The paired comparison applies the judge to both
3730
+ * sides of one conversation in one pass, so most of the judge's variance
3731
+ * cancels in its delta; an absolute mean carries that variance whole. What
3732
+ * ships here is the number, the difference against what serves today on the
3733
+ * same set, and the history. A person reads the trend.
3734
+ */
3735
+ export interface BenchmarkRun {
3736
+ id: string;
3737
+ benchmark_id: string;
3738
+ benchmark_name: string;
3739
+ run_id: string;
3740
+ workspace_id: string;
3741
+ incumbent_ref: EvaluationRef;
3742
+ candidate_ref: EvaluationRef;
3743
+ /** The set actually scored, recorded beside the score. */
3744
+ items_digest: string;
3745
+ rows_total: number;
3746
+ rows_scored: number;
3747
+ rows_failed: number;
3748
+ /** Absolute, on the frozen judge's scale, and null unless every row scored. */
3749
+ candidate_score: number | null;
3750
+ incumbent_score: number | null;
3751
+ score_delta: number | null;
3752
+ per_dimension: BenchmarkDimensionScore[];
3753
+ prompt_tokens: number;
3754
+ completion_tokens: number;
3755
+ spent_cents: number;
3756
+ ceiling_cents: number;
3757
+ status: BenchmarkRunStatus;
3758
+ /** The whole explanation when there is no score, so never empty on one. */
3759
+ status_reason: string | null;
3760
+ created_at: string;
3761
+ finished_at: string | null;
3762
+ }
3763
+ /**
3764
+ * One point on the trend line.
3765
+ *
3766
+ * It carries the checkpoint and what the person then decided, because a trend
3767
+ * with no idea what changed between two points is a chart rather than an
3768
+ * answer.
3769
+ */
3770
+ export interface BenchmarkHistoryPoint {
3771
+ benchmark_run_id: string;
3772
+ run_id: string;
3773
+ run_seq: number;
3774
+ rule_id: string | null;
3775
+ rule_name: string | null;
3776
+ created_at: string;
3777
+ candidate_score: number | null;
3778
+ incumbent_score: number | null;
3779
+ score_delta: number | null;
3780
+ checkpoint_id: string | null;
3781
+ decision: TrainingDecision | null;
3782
+ status: BenchmarkRunStatus;
3783
+ status_reason: string | null;
3784
+ }
3785
+ /** Where the conversations are copied FROM. Read once, at creation, never again. */
3786
+ export interface BenchmarkSource {
3787
+ kind: BenchmarkSourceKind;
3788
+ trace_ids?: string[];
3789
+ dataset_id?: string;
3790
+ split?: string;
3791
+ }
3792
+ /**
3793
+ * The pin: 10 to 200 conversations, and a judge with a rubric to measure them.
3794
+ *
3795
+ * A conversation with nothing to ask a model is refused rather than skipped.
3796
+ * Pinning 47 of the 50 somebody chose is the set being wrong from the first
3797
+ * day, and they would never find out.
3798
+ */
3799
+ export interface BenchmarkCreateParams {
3800
+ name: string;
3801
+ judge_id: string;
3802
+ source: BenchmarkSource;
3803
+ /** Absent uses the judge's own model, and absent that the workspace default. */
3804
+ judge_model?: string;
3805
+ /** What both models are given to answer in. A shorter answer is a different answer. */
3806
+ max_tokens?: number;
3807
+ per_run_ceiling_cents: number;
3808
+ }
3809
+ export interface BenchmarkListParams {
3810
+ status?: BenchmarkStatus;
3811
+ limit?: number;
3812
+ offset?: number;
3813
+ }
3814
+ export interface BenchmarkItemListParams {
3815
+ /** Capped at 100 by the service. */
3816
+ limit?: number;
3817
+ offset?: number;
3818
+ }
3819
+ export interface BenchmarkHistoryParams {
3820
+ limit?: number;
3821
+ offset?: number;
3822
+ }
3823
+ /** A reason is a courtesy here, not a requirement. */
3824
+ export interface BenchmarkRetireParams {
3825
+ reason?: string;
3826
+ }
3827
+ /**
3828
+ * Attach a benchmark to a rule, or detach it with a present null.
3829
+ *
3830
+ * Not optional, and not omittable: this route sets the field, so an absent key
3831
+ * would be a request with nothing in it. Send the id to attach, null to detach.
3832
+ */
3833
+ export interface TrainingRuleBenchmarkRequest {
3834
+ benchmark_id: string | null;
3835
+ }
3617
3836
  export interface TrainingRuleListResponse {
3618
3837
  rules: TrainingRule[];
3619
3838
  total: number;
@@ -3642,6 +3861,12 @@ export interface TrainingRunListResponse {
3642
3861
  export interface TrainingRunResponse {
3643
3862
  run: TrainingRun;
3644
3863
  timeline: TrainingRunEvent[];
3864
+ /**
3865
+ * The standing benchmark's two absolute numbers for this run, beside the
3866
+ * paired verdict and never in place of it. Null when no benchmark is
3867
+ * attached to the rule.
3868
+ */
3869
+ benchmark: BenchmarkRun | null;
3645
3870
  links: TrainingRunLinks;
3646
3871
  available_actions: TrainingRunActions;
3647
3872
  }
@@ -3666,3 +3891,17 @@ export interface InferenceAliasResponse {
3666
3891
  export interface InferenceAliasDeleteResponse {
3667
3892
  deleted: boolean;
3668
3893
  }
3894
+ export interface BenchmarkListResponse {
3895
+ benchmarks: Benchmark[];
3896
+ total: number;
3897
+ }
3898
+ export interface BenchmarkItemsResponse {
3899
+ items: BenchmarkItem[];
3900
+ total: number;
3901
+ }
3902
+ /** The trend, newest first. Every replay is a point, including the scoreless ones. */
3903
+ export interface BenchmarkHistoryResponse {
3904
+ benchmark_id: string;
3905
+ points: BenchmarkHistoryPoint[];
3906
+ total: number;
3907
+ }
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "runbios-sdk",
3
- "version": "0.2.1-rc.151",
3
+ "version": "0.2.1-rc.156",
4
4
  "description": "Official TypeScript SDK for the Run BiOS training and deployment platform API",
5
5
  "type": "module",
6
6
  "main": "./dist/index.js",