runbios-sdk 0.2.1-rc.151 → 0.2.1-rc.156
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/dist/index.d.ts +2 -2
- package/dist/index.js +1 -1
- package/dist/resources/loop.d.ts +98 -5
- package/dist/resources/loop.js +138 -4
- package/dist/types.d.ts +239 -0
- package/package.json +1 -1
package/dist/index.d.ts
CHANGED
|
@@ -36,7 +36,7 @@ import { Loop } from './resources/loop.js';
|
|
|
36
36
|
* SDK version. Sent as part of the User-Agent header.
|
|
37
37
|
* Must match package.json "version" -- enforced by a contract test.
|
|
38
38
|
*/
|
|
39
|
-
export declare const VERSION = "0.2.1-rc.
|
|
39
|
+
export declare const VERSION = "0.2.1-rc.156";
|
|
40
40
|
export declare class RunBiOS {
|
|
41
41
|
/** Search models, fetch configs, check adapter compatibility. */
|
|
42
42
|
readonly models: Models;
|
|
@@ -75,6 +75,6 @@ export { Training } from './resources/training.js';
|
|
|
75
75
|
export { Wallet } from './resources/wallet.js';
|
|
76
76
|
export { GPU, type GPURecommendation } from './resources/gpu.js';
|
|
77
77
|
export { Inference, validateChatRequest, parseSSE, buildInferenceRequest, contextSizingBasis, type ContextSizingBasis, CONTEXT_DEFAULT_CEILING, CONTEXT_EDITABLE_FLOOR, type ChatMessage, type FunctionTool, type ChatCompletionParams, type ChatCompletionResponse, type ChatCompletionChunk, } from './resources/inference.js';
|
|
78
|
-
export type { BiOSConfig, PaginatedResponse, ApiErrorBody, AvailableGpuAlternative, CapacityMinimumRequirement, InferenceBookingAccepted, Model, ModelSearchParams, ModelSearchResponse, ModelDetailResponse, ModelConfig, Dataset, DatasetListParams, DatasetListResponse, DatasetUploadParams, DatasetPreview, DatasetPreviewParams, DatasetImportHFParams, DatasetRegisterHFParams, DatasetHubSearchParams, DatasetHubPreviewParams, DatasetValidation, DatasetFormatVariant, DatasetFormatSpec, DatasetFormatSpecs, DatasetStorageUsage, TrainingMethod, RLHFAlgorithm, AdapterType, TrainingJobStatus, TrainingStatusFilter, TrainingCreateParams, TrainingListParams, TrainingListResponse, TrainingJob, TrainingMetrics, TrainingResourceMetrics, TrainingDeviceMetrics, MetricPoint, MetricGraphConfig, TrainingCheckpoint, TrainingLogs, TrainingLogEntry, TrainingStopResponse, TrainingResumeResponse, CanonicalTrainingRequest, TrainingPreflightDataset, TrainingPreflightWarning, TrainingPreflightResponse, TrainingCapabilityChoice, TrainingConfigFieldCapability, TrainingCapabilities, GPUChoice, WalletBalance, Transaction, TransactionListResponse, TransactionListParams, GPUInfo, GPUPricingResponse, GPUOptionsParams, TrainingRecommendParams, TrainingAdvisorVerdict, TrainingAdvisorRecommendation, TrainingAdvisorUserShape, TrainingAdvisorJustification, TrainingAdvisorBasis, GPUOption, GPUOptionSuggestion, GPUOptionsResponse, InferenceStatus, InferenceCreateParams, InferenceUpdateParams, InferenceDeployment, InferenceDeploymentSummary, InferenceCreateResponse, InferenceListResponse, InferenceUpdateResponse, InferenceLifecycleResponse, InferenceDeleteResponse, InferenceGPUOptionMarket, InferenceGPUOption, InferenceGPUAlternative, InferenceGPUOptionsParams, InferenceGPUOptionsResponse, InferenceCanonicalRequest, InferencePreflightWarning, InferencePreflightResponse, AdapterCompatibility, AdapterCompatibilityResponse, AdapterCompatibilityParams, SupportedArchitecture, ArchitectureScope, SupportedArchitecturesResponse, ApiKey, ApiKeyScope, ApiKeyIntrospection, Organization, OrgMember, OrgInvite, Workspace, Integration, IntegrationCreateParams, StorageUsage, StorageObject, PromptTokensDetails, ChatCompletionUsage, TrainingCadence, TrainingCombinator, ServingKind, TrainingRulePausedReason, LoopTrainingMethod, TrainType, GradersScope, TrainingTrigger, TrainingRunState, TrainingVerdict, TrainingDecision, NotifyState, EvaluationStatus, EvaluationItemStatus, EvaluationWinner, EvaluationWarningCode, ConsentVia, TrainingToolCall, TrainingMessage, TrainingRuleServing, TrainingGPURung, TrainingRule, TrainingRuleConsent, TrainingRulePreflightRefusal, TrainingRuleKeyRef, TrainingRuleKeyCheck, TrainingRulePreflight, TrainingRuleBuildSpec, TrainingRuleTriggerInput, TrainingRuleTrainingInput, TrainingRuleDeployInput, TrainingRuleMoneyInput, TrainingRuleEvaluationInput, TrainingRulePromotionInput, TrainingRuleAcceptTerms, TrainingRulePreflightRequest, TrainingRuleCreateRequest, TrainingRuleUpdateRequest, TrainingRuleConsentRequest, TrainingRuleListParams, TrainingRunPromoteRequest, TrainingRunRejectRequest, TrainingRunRollbackRequest, TrainingRunCancelRequest, TrainingRunListParams, TrainingRun, TrainingRunSummary, TrainingRunEvent, TrainingRunLinks, TrainingRunActions, EvaluationRef, EvaluationDecoding, EvaluationJudgeDimension, EvaluationDimensionScore, EvaluationGraderScore, EvaluationWarning, EvaluationMargin, JudgeAgreementDimension, JudgeAgreement, Evaluation, EvaluationItemSide, EvaluationItem, EvaluationItemListParams, JudgeAgreementParams, AgentSettings, AgentSettingsRequest, InferenceAlias, InferenceAliasRequest, TrainingRuleListResponse, TrainingRuleResponse, TrainingRuleMutationResponse, TrainingRuleDeleteResponse, TrainingRunListResponse, TrainingRunResponse, TrainingRunActionResponse, EvaluationItemsResponse, AgentSettingsResponse, InferenceAliasListResponse, InferenceAliasResponse, InferenceAliasDeleteResponse, } from './types.js';
|
|
78
|
+
export type { BiOSConfig, PaginatedResponse, ApiErrorBody, AvailableGpuAlternative, CapacityMinimumRequirement, InferenceBookingAccepted, Model, ModelSearchParams, ModelSearchResponse, ModelDetailResponse, ModelConfig, Dataset, DatasetListParams, DatasetListResponse, DatasetUploadParams, DatasetPreview, DatasetPreviewParams, DatasetImportHFParams, DatasetRegisterHFParams, DatasetHubSearchParams, DatasetHubPreviewParams, DatasetValidation, DatasetFormatVariant, DatasetFormatSpec, DatasetFormatSpecs, DatasetStorageUsage, TrainingMethod, RLHFAlgorithm, AdapterType, TrainingJobStatus, TrainingStatusFilter, TrainingCreateParams, TrainingListParams, TrainingListResponse, TrainingJob, TrainingMetrics, TrainingResourceMetrics, TrainingDeviceMetrics, MetricPoint, MetricGraphConfig, TrainingCheckpoint, TrainingLogs, TrainingLogEntry, TrainingStopResponse, TrainingResumeResponse, CanonicalTrainingRequest, TrainingPreflightDataset, TrainingPreflightWarning, TrainingPreflightResponse, TrainingCapabilityChoice, TrainingConfigFieldCapability, TrainingCapabilities, GPUChoice, WalletBalance, Transaction, TransactionListResponse, TransactionListParams, GPUInfo, GPUPricingResponse, GPUOptionsParams, TrainingRecommendParams, TrainingAdvisorVerdict, TrainingAdvisorRecommendation, TrainingAdvisorUserShape, TrainingAdvisorJustification, TrainingAdvisorBasis, GPUOption, GPUOptionSuggestion, GPUOptionsResponse, InferenceStatus, InferenceCreateParams, InferenceUpdateParams, InferenceDeployment, InferenceDeploymentSummary, InferenceCreateResponse, InferenceListResponse, InferenceUpdateResponse, InferenceLifecycleResponse, InferenceDeleteResponse, InferenceGPUOptionMarket, InferenceGPUOption, InferenceGPUAlternative, InferenceGPUOptionsParams, InferenceGPUOptionsResponse, InferenceCanonicalRequest, InferencePreflightWarning, InferencePreflightResponse, AdapterCompatibility, AdapterCompatibilityResponse, AdapterCompatibilityParams, SupportedArchitecture, ArchitectureScope, SupportedArchitecturesResponse, ApiKey, ApiKeyScope, ApiKeyIntrospection, Organization, OrgMember, OrgInvite, Workspace, Integration, IntegrationCreateParams, StorageUsage, StorageObject, PromptTokensDetails, ChatCompletionUsage, TrainingCadence, TrainingCombinator, ServingKind, TrainingRulePausedReason, LoopTrainingMethod, TrainType, GradersScope, TrainingTrigger, TrainingRunState, TrainingVerdict, TrainingDecision, NotifyState, EvaluationStatus, EvaluationItemStatus, EvaluationWinner, EvaluationWarningCode, ConsentVia, TrainingToolCall, TrainingMessage, TrainingRuleServing, TrainingGPURung, TrainingRule, TrainingRuleConsent, TrainingRulePreflightRefusal, TrainingRuleKeyRef, TrainingRuleKeyCheck, TrainingRulePreflight, TrainingRuleBuildSpec, TrainingRuleTriggerInput, TrainingRuleTrainingInput, TrainingRuleDeployInput, TrainingRuleMoneyInput, TrainingRuleEvaluationInput, TrainingRulePromotionInput, TrainingRuleAcceptTerms, TrainingRulePreflightRequest, TrainingRuleCreateRequest, TrainingRuleUpdateRequest, TrainingRuleConsentRequest, TrainingRuleListParams, TrainingRunPromoteRequest, TrainingRunRejectRequest, TrainingRunRollbackRequest, TrainingRunCancelRequest, TrainingRunListParams, TrainingRun, TrainingRunSummary, TrainingRunEvent, TrainingRunLinks, TrainingRunActions, EvaluationRef, EvaluationDecoding, EvaluationJudgeDimension, EvaluationDimensionScore, EvaluationGraderScore, EvaluationWarning, EvaluationMargin, JudgeAgreementDimension, JudgeAgreement, Evaluation, EvaluationItemSide, EvaluationItem, EvaluationItemListParams, JudgeAgreementParams, AgentSettings, AgentSettingsRequest, BenchmarkStatus, BenchmarkSourceKind, BenchmarkRunStatus, BenchmarkScoring, BenchmarkDimensionScore, Benchmark, BenchmarkItem, BenchmarkRun, BenchmarkHistoryPoint, BenchmarkSource, BenchmarkCreateParams, BenchmarkListParams, BenchmarkItemListParams, BenchmarkHistoryParams, BenchmarkRetireParams, TrainingRuleBenchmarkRequest, InferenceAlias, InferenceAliasRequest, TrainingRuleListResponse, TrainingRuleResponse, TrainingRuleMutationResponse, TrainingRuleDeleteResponse, TrainingRunListResponse, TrainingRunResponse, TrainingRunActionResponse, EvaluationItemsResponse, AgentSettingsResponse, InferenceAliasListResponse, InferenceAliasResponse, InferenceAliasDeleteResponse, BenchmarkListResponse, BenchmarkItemsResponse, BenchmarkHistoryResponse, } from './types.js';
|
|
79
79
|
/** The closed set of run states a training run never leaves. */
|
|
80
80
|
export { TERMINAL_RUN_STATES } from './types.js';
|
package/dist/index.js
CHANGED
|
@@ -36,7 +36,7 @@ import { Loop } from './resources/loop.js';
|
|
|
36
36
|
* SDK version. Sent as part of the User-Agent header.
|
|
37
37
|
* Must match package.json "version" -- enforced by a contract test.
|
|
38
38
|
*/
|
|
39
|
-
export const VERSION = '0.2.1-rc.
|
|
39
|
+
export const VERSION = '0.2.1-rc.156';
|
|
40
40
|
export class RunBiOS {
|
|
41
41
|
/** Search models, fetch configs, check adapter compatibility. */
|
|
42
42
|
models;
|
package/dist/resources/loop.d.ts
CHANGED
|
@@ -1,5 +1,5 @@
|
|
|
1
1
|
import type { HttpClient } from '../client.js';
|
|
2
|
-
import type { LoopTrace, LoopTraceListResponse, LoopTraceListParams, LoopCaptureParams, LoopCaptureResult, LoopImportParams, LoopImportResult, LoopSignal, LoopSignalParams, LoopCandidate, LoopCandidateParams, LoopGrader, LoopGraderParams, LoopLabel, LoopLabelCount, LoopGradeResult, LoopDataset, LoopDatasetListParams, LoopDatasetItem, LoopDatasetCreateParams, LoopConfig, LoopConfigParams, LoopBuildRule, LoopBuildRuleParams, LoopBuildRuleUpdateParams, LoopStats, LoopJudge, LoopJudgeParams, LoopAgentCredential, LoopAgentStatus, LoopSampleRun, LoopSampleRunParams, LoopJudgeRun, LoopJudgeRunItems, LoopJudgeRunItemStatus, LoopJudgeWork, LoopJudgeVerdict, LoopJudgeVerdictResult, TrainingRule, TrainingRulePreflight, TrainingRulePreflightRequest, TrainingRuleCreateRequest, TrainingRuleUpdateRequest, TrainingRuleConsentRequest, TrainingRuleListParams, TrainingRuleListResponse, TrainingRuleResponse, TrainingRuleMutationResponse, TrainingRuleDeleteResponse, TrainingRun, TrainingRunListParams, TrainingRunListResponse, TrainingRunResponse, TrainingRunActionResponse, TrainingRunPromoteRequest, TrainingRunRejectRequest, TrainingRunRollbackRequest, TrainingRunCancelRequest, Evaluation, EvaluationItemListParams, EvaluationItemsResponse, JudgeAgreement, JudgeAgreementParams, AgentSettings, AgentSettingsRequest } from '../types.js';
|
|
2
|
+
import type { LoopTrace, LoopTraceListResponse, LoopTraceListParams, LoopCaptureParams, LoopCaptureResult, LoopImportParams, LoopImportResult, LoopSignal, LoopSignalParams, LoopCandidate, LoopCandidateParams, LoopGrader, LoopGraderParams, LoopLabel, LoopLabelCount, LoopGradeResult, LoopDataset, LoopDatasetListParams, LoopDatasetItem, LoopDatasetCreateParams, LoopConfig, LoopConfigParams, LoopBuildRule, LoopBuildRuleParams, LoopBuildRuleUpdateParams, LoopStats, LoopJudge, LoopJudgeParams, LoopAgentCredential, LoopAgentStatus, LoopSampleRun, LoopSampleRunParams, LoopJudgeRun, LoopJudgeRunItems, LoopJudgeRunItemStatus, LoopJudgeWork, LoopJudgeVerdict, LoopJudgeVerdictResult, TrainingRule, TrainingRulePreflight, TrainingRulePreflightRequest, TrainingRuleCreateRequest, TrainingRuleUpdateRequest, TrainingRuleConsentRequest, TrainingRuleListParams, TrainingRuleListResponse, TrainingRuleResponse, TrainingRuleMutationResponse, TrainingRuleDeleteResponse, TrainingRun, TrainingRunListParams, TrainingRunListResponse, TrainingRunResponse, TrainingRunActionResponse, TrainingRunPromoteRequest, TrainingRunRejectRequest, TrainingRunRollbackRequest, TrainingRunCancelRequest, Evaluation, EvaluationItemListParams, EvaluationItemsResponse, JudgeAgreement, JudgeAgreementParams, Benchmark, BenchmarkCreateParams, BenchmarkHistoryParams, BenchmarkHistoryResponse, BenchmarkItemListParams, BenchmarkItemsResponse, BenchmarkListParams, BenchmarkListResponse, BenchmarkRetireParams, BenchmarkRun, TrainingRuleBenchmarkRequest, AgentSettings, AgentSettingsRequest } from '../types.js';
|
|
3
3
|
/**
|
|
4
4
|
* The Conscious Loop -- capture what your model was asked and answered, record
|
|
5
5
|
* whether it was right, and turn those judgements into training data.
|
|
@@ -567,10 +567,11 @@ export declare class Loop {
|
|
|
567
567
|
* `valid: false` with an EMPTY `refusals` list is not nothing: check
|
|
568
568
|
* `unreachable`, which names every peer the platform could not reach. The
|
|
569
569
|
* rule is savable in that state and would be paused, but the estimate around
|
|
570
|
-
* it is not trustworthy -- both `worst_hourly_*` are `0
|
|
571
|
-
*
|
|
572
|
-
*
|
|
573
|
-
* `unreachable.length > 0`
|
|
570
|
+
* it is not trustworthy -- both `worst_hourly_*` are `0`, and so are both
|
|
571
|
+
* `max_*_hours` beside them -- so do not quote those figures to anyone.
|
|
572
|
+
* `model_revision` is empty unless training-service actually pinned a
|
|
573
|
+
* commit; test `unreachable.length > 0` to tell a degraded pass from a
|
|
574
|
+
* refusal.
|
|
574
575
|
*/
|
|
575
576
|
preflightTrainingRule(params: TrainingRulePreflightRequest): Promise<TrainingRulePreflight>;
|
|
576
577
|
/**
|
|
@@ -739,6 +740,98 @@ export declare class Loop {
|
|
|
739
740
|
* reviewer overlap yet" is an answer, and 100% of two pairs is not.
|
|
740
741
|
*/
|
|
741
742
|
getJudgeAgreement(judgeId: string, params?: JudgeAgreementParams): Promise<JudgeAgreement>;
|
|
743
|
+
/**
|
|
744
|
+
* Pin a fixed set of conversations, with a judge frozen beside them, and
|
|
745
|
+
* measure every future run against it.
|
|
746
|
+
*
|
|
747
|
+
* The comparison answers "is this candidate better than what serves today".
|
|
748
|
+
* It cannot answer "is my model getting better", because its rows, its
|
|
749
|
+
* opponent and its judge all move between runs. A benchmark is the other
|
|
750
|
+
* instrument: the same conversations, the same oracle, the same decoding,
|
|
751
|
+
* replayed against both models on every run that finishes its comparison,
|
|
752
|
+
* and reported as two absolute numbers on a scale that does not move.
|
|
753
|
+
*
|
|
754
|
+
* 10 to 200 conversations. A named conversation with nothing to ask a model
|
|
755
|
+
* is refused rather than skipped, because pinning 47 of the 50 you chose is
|
|
756
|
+
* the set being wrong from the first day and you would never find out.
|
|
757
|
+
*
|
|
758
|
+
* It raises no amount you have already agreed to. The replay's calls come
|
|
759
|
+
* out of the rule's existing `eval_ceiling_cents`, and
|
|
760
|
+
* `per_run_ceiling_cents` can only lower what is spent inside that.
|
|
761
|
+
*/
|
|
762
|
+
createBenchmark(params: BenchmarkCreateParams): Promise<Benchmark>;
|
|
763
|
+
/** Your benchmarks, newest first. Filter by `status` to hide retired ones. */
|
|
764
|
+
listBenchmarks(params?: BenchmarkListParams): Promise<BenchmarkListResponse>;
|
|
765
|
+
/**
|
|
766
|
+
* One benchmark: the frozen oracle, the scoring rule and the fingerprint of
|
|
767
|
+
* the set.
|
|
768
|
+
*
|
|
769
|
+
* `items_digest` is recomputed from the rows and compared at the start of
|
|
770
|
+
* every replay, so "the set cannot drift" is something you can check rather
|
|
771
|
+
* than something the platform promises.
|
|
772
|
+
*/
|
|
773
|
+
getBenchmark(id: string): Promise<Benchmark>;
|
|
774
|
+
/**
|
|
775
|
+
* The pinned conversations, paged. `limit` is capped at 100 by the service.
|
|
776
|
+
*
|
|
777
|
+
* `source_trace_id` and `source_trace_url` come back null once the
|
|
778
|
+
* conversation a row was copied from has been deleted. The row itself stays
|
|
779
|
+
* and every number already measured against it stays exactly as comparable
|
|
780
|
+
* as it was: retention cannot shrink a benchmark.
|
|
781
|
+
*/
|
|
782
|
+
listBenchmarkItems(id: string, params?: BenchmarkItemListParams): Promise<BenchmarkItemsResponse>;
|
|
783
|
+
/**
|
|
784
|
+
* The trend: every score this benchmark has produced, newest first, each
|
|
785
|
+
* beside the checkpoint that produced it and what you then did about it.
|
|
786
|
+
*
|
|
787
|
+
* EVERY REPLAY IS A POINT, including the ones with no numbers on them. A
|
|
788
|
+
* series that silently dropped the runs where the benchmark could not be
|
|
789
|
+
* scored would read as an unbroken line and not be one, so those points come
|
|
790
|
+
* back with `candidate_score` null and `status_reason` saying why.
|
|
791
|
+
*/
|
|
792
|
+
getBenchmarkHistory(id: string, params?: BenchmarkHistoryParams): Promise<BenchmarkHistoryResponse>;
|
|
793
|
+
/**
|
|
794
|
+
* Stop replaying a benchmark, and detach it from every rule that names it.
|
|
795
|
+
*
|
|
796
|
+
* Nothing measured is removed. There is no delete and no update on purpose:
|
|
797
|
+
* the series of numbers measured against a set is what a benchmark is for,
|
|
798
|
+
* so an edit would make everything before it incomparable with everything
|
|
799
|
+
* after it and a delete would throw the series away. Changing the set means
|
|
800
|
+
* pinning a new benchmark, and retiring the old one releases its name.
|
|
801
|
+
*
|
|
802
|
+
* Retiring twice is somebody pressing a button twice: the second call
|
|
803
|
+
* answers with the retired benchmark rather than a refusal.
|
|
804
|
+
*/
|
|
805
|
+
retireBenchmark(id: string, params?: BenchmarkRetireParams): Promise<Benchmark>;
|
|
806
|
+
/**
|
|
807
|
+
* One replay in full: both absolute scores, the difference between them on
|
|
808
|
+
* the same set, the per-dimension breakdown and what it cost.
|
|
809
|
+
*
|
|
810
|
+
* Read `status` before you read the scores. A replay that did not score
|
|
811
|
+
* every pinned conversation publishes no score at all -- all three of
|
|
812
|
+
* `candidate_score`, `incumbent_score` and `score_delta` are null and
|
|
813
|
+
* `status_reason` is the sentence that explains it -- because a mean over
|
|
814
|
+
* whichever conversations happened to succeed is a measurement of a
|
|
815
|
+
* different set, which is the exact defect a standing benchmark removes.
|
|
816
|
+
*/
|
|
817
|
+
getBenchmarkRun(id: string): Promise<BenchmarkRun>;
|
|
818
|
+
/**
|
|
819
|
+
* Attach a benchmark to a rule, or send `{ benchmark_id: null }` to detach
|
|
820
|
+
* it.
|
|
821
|
+
*
|
|
822
|
+
* Its own route rather than a field on the rule body, because it is a
|
|
823
|
+
* decision to replay a fixed set on every future run of this rule for as
|
|
824
|
+
* long as it stands, and its refusals -- retired, or belonging to another
|
|
825
|
+
* workspace -- are about the benchmark rather than about the rule.
|
|
826
|
+
*
|
|
827
|
+
* It does not invalidate consent and the reply says so: attaching raises
|
|
828
|
+
* neither the amount set aside for judge calls nor the amount set aside for
|
|
829
|
+
* keeping the new model available, so nobody is asked to read the same
|
|
830
|
+
* sentence again. A retired benchmark is refused with `BENCHMARK_RETIRED`,
|
|
831
|
+
* because a rule pointed at one would report no number on every run and say
|
|
832
|
+
* nothing about why.
|
|
833
|
+
*/
|
|
834
|
+
setTrainingRuleBenchmark(id: string, params: TrainingRuleBenchmarkRequest): Promise<TrainingRuleMutationResponse>;
|
|
742
835
|
/**
|
|
743
836
|
* The workspace's agent options: which model it defaults to, the system
|
|
744
837
|
* prompts it judges and samples with, and the monthly cap on what its model
|
package/dist/resources/loop.js
CHANGED
|
@@ -776,10 +776,11 @@ export class Loop {
|
|
|
776
776
|
* `valid: false` with an EMPTY `refusals` list is not nothing: check
|
|
777
777
|
* `unreachable`, which names every peer the platform could not reach. The
|
|
778
778
|
* rule is savable in that state and would be paused, but the estimate around
|
|
779
|
-
* it is not trustworthy -- both `worst_hourly_*` are `0
|
|
780
|
-
*
|
|
781
|
-
*
|
|
782
|
-
* `unreachable.length > 0`
|
|
779
|
+
* it is not trustworthy -- both `worst_hourly_*` are `0`, and so are both
|
|
780
|
+
* `max_*_hours` beside them -- so do not quote those figures to anyone.
|
|
781
|
+
* `model_revision` is empty unless training-service actually pinned a
|
|
782
|
+
* commit; test `unreachable.length > 0` to tell a degraded pass from a
|
|
783
|
+
* refusal.
|
|
783
784
|
*/
|
|
784
785
|
async preflightTrainingRule(params) {
|
|
785
786
|
return this._http.fetchPost('/api/loop/training-rules/preflight', params);
|
|
@@ -1030,6 +1031,139 @@ export class Loop {
|
|
|
1030
1031
|
const qs = q.toString();
|
|
1031
1032
|
return this._http.fetchGet(`/api/loop/judges/${encodeURIComponent(judgeId)}/agreement${qs ? `?${qs}` : ''}`);
|
|
1032
1033
|
}
|
|
1034
|
+
// ── the standing benchmark ────────────────────────────────────────────
|
|
1035
|
+
/**
|
|
1036
|
+
* Pin a fixed set of conversations, with a judge frozen beside them, and
|
|
1037
|
+
* measure every future run against it.
|
|
1038
|
+
*
|
|
1039
|
+
* The comparison answers "is this candidate better than what serves today".
|
|
1040
|
+
* It cannot answer "is my model getting better", because its rows, its
|
|
1041
|
+
* opponent and its judge all move between runs. A benchmark is the other
|
|
1042
|
+
* instrument: the same conversations, the same oracle, the same decoding,
|
|
1043
|
+
* replayed against both models on every run that finishes its comparison,
|
|
1044
|
+
* and reported as two absolute numbers on a scale that does not move.
|
|
1045
|
+
*
|
|
1046
|
+
* 10 to 200 conversations. A named conversation with nothing to ask a model
|
|
1047
|
+
* is refused rather than skipped, because pinning 47 of the 50 you chose is
|
|
1048
|
+
* the set being wrong from the first day and you would never find out.
|
|
1049
|
+
*
|
|
1050
|
+
* It raises no amount you have already agreed to. The replay's calls come
|
|
1051
|
+
* out of the rule's existing `eval_ceiling_cents`, and
|
|
1052
|
+
* `per_run_ceiling_cents` can only lower what is spent inside that.
|
|
1053
|
+
*/
|
|
1054
|
+
async createBenchmark(params) {
|
|
1055
|
+
const res = await this._http.fetchPost('/api/loop/benchmarks', params);
|
|
1056
|
+
return res.benchmark;
|
|
1057
|
+
}
|
|
1058
|
+
/** Your benchmarks, newest first. Filter by `status` to hide retired ones. */
|
|
1059
|
+
async listBenchmarks(params = {}) {
|
|
1060
|
+
const q = new URLSearchParams();
|
|
1061
|
+
if (params.status)
|
|
1062
|
+
q.set('status', params.status);
|
|
1063
|
+
if (params.limit != null)
|
|
1064
|
+
q.set('limit', String(params.limit));
|
|
1065
|
+
if (params.offset != null)
|
|
1066
|
+
q.set('offset', String(params.offset));
|
|
1067
|
+
const qs = q.toString();
|
|
1068
|
+
return this._http.fetchGet(`/api/loop/benchmarks${qs ? `?${qs}` : ''}`);
|
|
1069
|
+
}
|
|
1070
|
+
/**
|
|
1071
|
+
* One benchmark: the frozen oracle, the scoring rule and the fingerprint of
|
|
1072
|
+
* the set.
|
|
1073
|
+
*
|
|
1074
|
+
* `items_digest` is recomputed from the rows and compared at the start of
|
|
1075
|
+
* every replay, so "the set cannot drift" is something you can check rather
|
|
1076
|
+
* than something the platform promises.
|
|
1077
|
+
*/
|
|
1078
|
+
async getBenchmark(id) {
|
|
1079
|
+
const res = await this._http.fetchGet(`/api/loop/benchmarks/${encodeURIComponent(id)}`);
|
|
1080
|
+
return res.benchmark;
|
|
1081
|
+
}
|
|
1082
|
+
/**
|
|
1083
|
+
* The pinned conversations, paged. `limit` is capped at 100 by the service.
|
|
1084
|
+
*
|
|
1085
|
+
* `source_trace_id` and `source_trace_url` come back null once the
|
|
1086
|
+
* conversation a row was copied from has been deleted. The row itself stays
|
|
1087
|
+
* and every number already measured against it stays exactly as comparable
|
|
1088
|
+
* as it was: retention cannot shrink a benchmark.
|
|
1089
|
+
*/
|
|
1090
|
+
async listBenchmarkItems(id, params = {}) {
|
|
1091
|
+
const q = new URLSearchParams();
|
|
1092
|
+
if (params.limit != null)
|
|
1093
|
+
q.set('limit', String(params.limit));
|
|
1094
|
+
if (params.offset != null)
|
|
1095
|
+
q.set('offset', String(params.offset));
|
|
1096
|
+
const qs = q.toString();
|
|
1097
|
+
return this._http.fetchGet(`/api/loop/benchmarks/${encodeURIComponent(id)}/items${qs ? `?${qs}` : ''}`);
|
|
1098
|
+
}
|
|
1099
|
+
/**
|
|
1100
|
+
* The trend: every score this benchmark has produced, newest first, each
|
|
1101
|
+
* beside the checkpoint that produced it and what you then did about it.
|
|
1102
|
+
*
|
|
1103
|
+
* EVERY REPLAY IS A POINT, including the ones with no numbers on them. A
|
|
1104
|
+
* series that silently dropped the runs where the benchmark could not be
|
|
1105
|
+
* scored would read as an unbroken line and not be one, so those points come
|
|
1106
|
+
* back with `candidate_score` null and `status_reason` saying why.
|
|
1107
|
+
*/
|
|
1108
|
+
async getBenchmarkHistory(id, params = {}) {
|
|
1109
|
+
const q = new URLSearchParams();
|
|
1110
|
+
if (params.limit != null)
|
|
1111
|
+
q.set('limit', String(params.limit));
|
|
1112
|
+
if (params.offset != null)
|
|
1113
|
+
q.set('offset', String(params.offset));
|
|
1114
|
+
const qs = q.toString();
|
|
1115
|
+
return this._http.fetchGet(`/api/loop/benchmarks/${encodeURIComponent(id)}/history${qs ? `?${qs}` : ''}`);
|
|
1116
|
+
}
|
|
1117
|
+
/**
|
|
1118
|
+
* Stop replaying a benchmark, and detach it from every rule that names it.
|
|
1119
|
+
*
|
|
1120
|
+
* Nothing measured is removed. There is no delete and no update on purpose:
|
|
1121
|
+
* the series of numbers measured against a set is what a benchmark is for,
|
|
1122
|
+
* so an edit would make everything before it incomparable with everything
|
|
1123
|
+
* after it and a delete would throw the series away. Changing the set means
|
|
1124
|
+
* pinning a new benchmark, and retiring the old one releases its name.
|
|
1125
|
+
*
|
|
1126
|
+
* Retiring twice is somebody pressing a button twice: the second call
|
|
1127
|
+
* answers with the retired benchmark rather than a refusal.
|
|
1128
|
+
*/
|
|
1129
|
+
async retireBenchmark(id, params = {}) {
|
|
1130
|
+
const res = await this._http.fetchPost(`/api/loop/benchmarks/${encodeURIComponent(id)}/retire`, params);
|
|
1131
|
+
return res.benchmark;
|
|
1132
|
+
}
|
|
1133
|
+
/**
|
|
1134
|
+
* One replay in full: both absolute scores, the difference between them on
|
|
1135
|
+
* the same set, the per-dimension breakdown and what it cost.
|
|
1136
|
+
*
|
|
1137
|
+
* Read `status` before you read the scores. A replay that did not score
|
|
1138
|
+
* every pinned conversation publishes no score at all -- all three of
|
|
1139
|
+
* `candidate_score`, `incumbent_score` and `score_delta` are null and
|
|
1140
|
+
* `status_reason` is the sentence that explains it -- because a mean over
|
|
1141
|
+
* whichever conversations happened to succeed is a measurement of a
|
|
1142
|
+
* different set, which is the exact defect a standing benchmark removes.
|
|
1143
|
+
*/
|
|
1144
|
+
async getBenchmarkRun(id) {
|
|
1145
|
+
const res = await this._http.fetchGet(`/api/loop/benchmark-runs/${encodeURIComponent(id)}`);
|
|
1146
|
+
return res.benchmark_run;
|
|
1147
|
+
}
|
|
1148
|
+
/**
|
|
1149
|
+
* Attach a benchmark to a rule, or send `{ benchmark_id: null }` to detach
|
|
1150
|
+
* it.
|
|
1151
|
+
*
|
|
1152
|
+
* Its own route rather than a field on the rule body, because it is a
|
|
1153
|
+
* decision to replay a fixed set on every future run of this rule for as
|
|
1154
|
+
* long as it stands, and its refusals -- retired, or belonging to another
|
|
1155
|
+
* workspace -- are about the benchmark rather than about the rule.
|
|
1156
|
+
*
|
|
1157
|
+
* It does not invalidate consent and the reply says so: attaching raises
|
|
1158
|
+
* neither the amount set aside for judge calls nor the amount set aside for
|
|
1159
|
+
* keeping the new model available, so nobody is asked to read the same
|
|
1160
|
+
* sentence again. A retired benchmark is refused with `BENCHMARK_RETIRED`,
|
|
1161
|
+
* because a rule pointed at one would report no number on every run and say
|
|
1162
|
+
* nothing about why.
|
|
1163
|
+
*/
|
|
1164
|
+
async setTrainingRuleBenchmark(id, params) {
|
|
1165
|
+
return this._http.fetchPut(`/api/loop/training-rules/${encodeURIComponent(id)}/benchmark`, params);
|
|
1166
|
+
}
|
|
1033
1167
|
// ── agent settings ────────────────────────────────────────────────────
|
|
1034
1168
|
/**
|
|
1035
1169
|
* The workspace's agent options: which model it defaults to, the system
|
package/dist/types.d.ts
CHANGED
|
@@ -3083,6 +3083,15 @@ export interface TrainingRule {
|
|
|
3083
3083
|
eval_max_rows: number;
|
|
3084
3084
|
eval_max_tokens: number;
|
|
3085
3085
|
min_holdout_rows: number;
|
|
3086
|
+
/**
|
|
3087
|
+
* The standing benchmark replayed on every run of this rule, beside the
|
|
3088
|
+
* per-run comparison and never instead of it. Null is the ordinary state.
|
|
3089
|
+
*
|
|
3090
|
+
* Set with `setTrainingRuleBenchmark`, not by an update to the rule: it is a
|
|
3091
|
+
* decision to replay a fixed set on every future run, and it has refusals of
|
|
3092
|
+
* its own. It raises no amount, so it does not invalidate consent.
|
|
3093
|
+
*/
|
|
3094
|
+
benchmark_id: string | null;
|
|
3086
3095
|
auto_promote: boolean;
|
|
3087
3096
|
promote_margin: number;
|
|
3088
3097
|
promote_min_win_rate: number;
|
|
@@ -3352,6 +3361,15 @@ export interface TrainingRun {
|
|
|
3352
3361
|
candidate_deadline_at: string | null;
|
|
3353
3362
|
candidate_cleanup_at: string | null;
|
|
3354
3363
|
evaluation_id: string | null;
|
|
3364
|
+
/** This run's replay of the rule's standing benchmark, if it had one. */
|
|
3365
|
+
benchmark_run_id: string | null;
|
|
3366
|
+
/**
|
|
3367
|
+
* What the replay's calls cost. Kept apart from `spent_eval_cents` so a
|
|
3368
|
+
* report can say what the comparison cost and what the benchmark cost; both
|
|
3369
|
+
* come out of `eval_ceiling_cents` and their sum can never exceed it, so
|
|
3370
|
+
* anything totalling what a run cost has to add this one too.
|
|
3371
|
+
*/
|
|
3372
|
+
benchmark_spent_cents: number;
|
|
3355
3373
|
alias_name: string | null;
|
|
3356
3374
|
alias_written_at: string | null;
|
|
3357
3375
|
serving_before: TrainingRuleServing | null;
|
|
@@ -3389,6 +3407,8 @@ export interface TrainingRunSummary {
|
|
|
3389
3407
|
billed_training_cents: number;
|
|
3390
3408
|
billed_candidate_cents: number;
|
|
3391
3409
|
spent_eval_cents: number;
|
|
3410
|
+
/** The fourth money column. A row that leaves it out adds up short. */
|
|
3411
|
+
benchmark_spent_cents: number;
|
|
3392
3412
|
training_job_id: string | null;
|
|
3393
3413
|
candidate_deployment_id: string | null;
|
|
3394
3414
|
evaluation_id: string | null;
|
|
@@ -3414,6 +3434,8 @@ export interface TrainingRunLinks {
|
|
|
3414
3434
|
candidate_url: string | null;
|
|
3415
3435
|
dataset_url: string | null;
|
|
3416
3436
|
evaluation_url: string | null;
|
|
3437
|
+
/** The TREND the benchmark number belongs to, not the one replay. */
|
|
3438
|
+
benchmark_url: string | null;
|
|
3417
3439
|
}
|
|
3418
3440
|
/** What this reader may do right now. A button that cannot work is never shown. */
|
|
3419
3441
|
export interface TrainingRunActions {
|
|
@@ -3614,6 +3636,203 @@ export interface InferenceAliasRequest {
|
|
|
3614
3636
|
target_inference_id: string;
|
|
3615
3637
|
origin?: string;
|
|
3616
3638
|
}
|
|
3639
|
+
/**
|
|
3640
|
+
* A benchmark is active until it is retired. There is no delete and no update:
|
|
3641
|
+
* the series of numbers measured against a set is what a benchmark is for, so
|
|
3642
|
+
* an edit would make every number before it incomparable with every number
|
|
3643
|
+
* after it, and a delete throws the series away.
|
|
3644
|
+
*/
|
|
3645
|
+
export type BenchmarkStatus = 'active' | 'retired';
|
|
3646
|
+
/** Where the pinned conversations are copied from. Read once, at creation. */
|
|
3647
|
+
export type BenchmarkSourceKind = 'traces' | 'dataset';
|
|
3648
|
+
/**
|
|
3649
|
+
* The life of one replay.
|
|
3650
|
+
*
|
|
3651
|
+
* `budget_stopped` reached the amount left for it inside the run's own ceiling,
|
|
3652
|
+
* and `abandoned` ran out of the time the run sets aside for it. Both are
|
|
3653
|
+
* points on the trend carrying `status_reason`, never silent gaps.
|
|
3654
|
+
*/
|
|
3655
|
+
export type BenchmarkRunStatus = 'open' | 'done' | 'failed' | 'budget_stopped' | 'abandoned';
|
|
3656
|
+
/**
|
|
3657
|
+
* How the pinned conversations become one number, frozen on the benchmark.
|
|
3658
|
+
*
|
|
3659
|
+
* `require_all_rows` is what makes the number comparable at all: a mean over
|
|
3660
|
+
* whichever conversations happened to succeed is a measurement of a different
|
|
3661
|
+
* set, so a short replay publishes no score and says why. A partial benchmark
|
|
3662
|
+
* is a missing number, never a lower one.
|
|
3663
|
+
*/
|
|
3664
|
+
export interface BenchmarkScoring {
|
|
3665
|
+
metric: string;
|
|
3666
|
+
scale: string;
|
|
3667
|
+
aggregate: string;
|
|
3668
|
+
require_all_rows: boolean;
|
|
3669
|
+
}
|
|
3670
|
+
/**
|
|
3671
|
+
* One rubric dimension of one replay, for both models.
|
|
3672
|
+
*
|
|
3673
|
+
* Deliberately not `EvaluationDimensionScore`: these are absolute means on a
|
|
3674
|
+
* fixed set and those are paired means on that run's own held-back rows.
|
|
3675
|
+
*/
|
|
3676
|
+
export interface BenchmarkDimensionScore {
|
|
3677
|
+
dimension: string;
|
|
3678
|
+
incumbent: number;
|
|
3679
|
+
candidate: number;
|
|
3680
|
+
delta: number;
|
|
3681
|
+
}
|
|
3682
|
+
/**
|
|
3683
|
+
* The frozen definition: the oracle, the scoring rule and the fingerprint of
|
|
3684
|
+
* the set. The conversations themselves are their own paged route.
|
|
3685
|
+
*/
|
|
3686
|
+
export interface Benchmark {
|
|
3687
|
+
id: string;
|
|
3688
|
+
workspace_id: string;
|
|
3689
|
+
name: string;
|
|
3690
|
+
status: BenchmarkStatus;
|
|
3691
|
+
/** Provenance only: everything needed from the judge is copied below it. */
|
|
3692
|
+
judge_id: string | null;
|
|
3693
|
+
judge_name: string;
|
|
3694
|
+
judge_model: string;
|
|
3695
|
+
judge_instructions: string;
|
|
3696
|
+
judge_dimensions: EvaluationJudgeDimension[];
|
|
3697
|
+
judge_system_prompt: string | null;
|
|
3698
|
+
decoding: EvaluationDecoding;
|
|
3699
|
+
scoring: BenchmarkScoring;
|
|
3700
|
+
item_count: number;
|
|
3701
|
+
/** Recomputed from the rows and compared at the start of every replay. */
|
|
3702
|
+
items_digest: string;
|
|
3703
|
+
/** A cap INSIDE the run's own amount for judge calls, never an addition to it. */
|
|
3704
|
+
per_run_ceiling_cents: number;
|
|
3705
|
+
created_by: string;
|
|
3706
|
+
created_at: string;
|
|
3707
|
+
retired_at: string | null;
|
|
3708
|
+
retired_by: string | null;
|
|
3709
|
+
}
|
|
3710
|
+
/**
|
|
3711
|
+
* One pinned conversation.
|
|
3712
|
+
*
|
|
3713
|
+
* `source_trace_id` is a link and is allowed to go null: the prompt was copied
|
|
3714
|
+
* at pin time, so retention removing the conversation takes away the ability to
|
|
3715
|
+
* open it and takes away nothing else. Every number already measured stays
|
|
3716
|
+
* exactly as comparable as it was.
|
|
3717
|
+
*/
|
|
3718
|
+
export interface BenchmarkItem {
|
|
3719
|
+
id: number;
|
|
3720
|
+
ordinal: number;
|
|
3721
|
+
source_trace_id: string | null;
|
|
3722
|
+
source_trace_url: string | null;
|
|
3723
|
+
prompt: TrainingMessage[];
|
|
3724
|
+
tools: unknown[] | null;
|
|
3725
|
+
}
|
|
3726
|
+
/**
|
|
3727
|
+
* One replay: this benchmark, on this training run, against both models.
|
|
3728
|
+
*
|
|
3729
|
+
* IT DOES NOT GATE PROMOTION. The paired comparison applies the judge to both
|
|
3730
|
+
* sides of one conversation in one pass, so most of the judge's variance
|
|
3731
|
+
* cancels in its delta; an absolute mean carries that variance whole. What
|
|
3732
|
+
* ships here is the number, the difference against what serves today on the
|
|
3733
|
+
* same set, and the history. A person reads the trend.
|
|
3734
|
+
*/
|
|
3735
|
+
export interface BenchmarkRun {
|
|
3736
|
+
id: string;
|
|
3737
|
+
benchmark_id: string;
|
|
3738
|
+
benchmark_name: string;
|
|
3739
|
+
run_id: string;
|
|
3740
|
+
workspace_id: string;
|
|
3741
|
+
incumbent_ref: EvaluationRef;
|
|
3742
|
+
candidate_ref: EvaluationRef;
|
|
3743
|
+
/** The set actually scored, recorded beside the score. */
|
|
3744
|
+
items_digest: string;
|
|
3745
|
+
rows_total: number;
|
|
3746
|
+
rows_scored: number;
|
|
3747
|
+
rows_failed: number;
|
|
3748
|
+
/** Absolute, on the frozen judge's scale, and null unless every row scored. */
|
|
3749
|
+
candidate_score: number | null;
|
|
3750
|
+
incumbent_score: number | null;
|
|
3751
|
+
score_delta: number | null;
|
|
3752
|
+
per_dimension: BenchmarkDimensionScore[];
|
|
3753
|
+
prompt_tokens: number;
|
|
3754
|
+
completion_tokens: number;
|
|
3755
|
+
spent_cents: number;
|
|
3756
|
+
ceiling_cents: number;
|
|
3757
|
+
status: BenchmarkRunStatus;
|
|
3758
|
+
/** The whole explanation when there is no score, so never empty on one. */
|
|
3759
|
+
status_reason: string | null;
|
|
3760
|
+
created_at: string;
|
|
3761
|
+
finished_at: string | null;
|
|
3762
|
+
}
|
|
3763
|
+
/**
|
|
3764
|
+
* One point on the trend line.
|
|
3765
|
+
*
|
|
3766
|
+
* It carries the checkpoint and what the person then decided, because a trend
|
|
3767
|
+
* with no idea what changed between two points is a chart rather than an
|
|
3768
|
+
* answer.
|
|
3769
|
+
*/
|
|
3770
|
+
export interface BenchmarkHistoryPoint {
|
|
3771
|
+
benchmark_run_id: string;
|
|
3772
|
+
run_id: string;
|
|
3773
|
+
run_seq: number;
|
|
3774
|
+
rule_id: string | null;
|
|
3775
|
+
rule_name: string | null;
|
|
3776
|
+
created_at: string;
|
|
3777
|
+
candidate_score: number | null;
|
|
3778
|
+
incumbent_score: number | null;
|
|
3779
|
+
score_delta: number | null;
|
|
3780
|
+
checkpoint_id: string | null;
|
|
3781
|
+
decision: TrainingDecision | null;
|
|
3782
|
+
status: BenchmarkRunStatus;
|
|
3783
|
+
status_reason: string | null;
|
|
3784
|
+
}
|
|
3785
|
+
/** Where the conversations are copied FROM. Read once, at creation, never again. */
|
|
3786
|
+
export interface BenchmarkSource {
|
|
3787
|
+
kind: BenchmarkSourceKind;
|
|
3788
|
+
trace_ids?: string[];
|
|
3789
|
+
dataset_id?: string;
|
|
3790
|
+
split?: string;
|
|
3791
|
+
}
|
|
3792
|
+
/**
|
|
3793
|
+
* The pin: 10 to 200 conversations, and a judge with a rubric to measure them.
|
|
3794
|
+
*
|
|
3795
|
+
* A conversation with nothing to ask a model is refused rather than skipped.
|
|
3796
|
+
* Pinning 47 of the 50 somebody chose is the set being wrong from the first
|
|
3797
|
+
* day, and they would never find out.
|
|
3798
|
+
*/
|
|
3799
|
+
export interface BenchmarkCreateParams {
|
|
3800
|
+
name: string;
|
|
3801
|
+
judge_id: string;
|
|
3802
|
+
source: BenchmarkSource;
|
|
3803
|
+
/** Absent uses the judge's own model, and absent that the workspace default. */
|
|
3804
|
+
judge_model?: string;
|
|
3805
|
+
/** What both models are given to answer in. A shorter answer is a different answer. */
|
|
3806
|
+
max_tokens?: number;
|
|
3807
|
+
per_run_ceiling_cents: number;
|
|
3808
|
+
}
|
|
3809
|
+
export interface BenchmarkListParams {
|
|
3810
|
+
status?: BenchmarkStatus;
|
|
3811
|
+
limit?: number;
|
|
3812
|
+
offset?: number;
|
|
3813
|
+
}
|
|
3814
|
+
export interface BenchmarkItemListParams {
|
|
3815
|
+
/** Capped at 100 by the service. */
|
|
3816
|
+
limit?: number;
|
|
3817
|
+
offset?: number;
|
|
3818
|
+
}
|
|
3819
|
+
export interface BenchmarkHistoryParams {
|
|
3820
|
+
limit?: number;
|
|
3821
|
+
offset?: number;
|
|
3822
|
+
}
|
|
3823
|
+
/** A reason is a courtesy here, not a requirement. */
|
|
3824
|
+
export interface BenchmarkRetireParams {
|
|
3825
|
+
reason?: string;
|
|
3826
|
+
}
|
|
3827
|
+
/**
|
|
3828
|
+
* Attach a benchmark to a rule, or detach it with a present null.
|
|
3829
|
+
*
|
|
3830
|
+
* Not optional, and not omittable: this route sets the field, so an absent key
|
|
3831
|
+
* would be a request with nothing in it. Send the id to attach, null to detach.
|
|
3832
|
+
*/
|
|
3833
|
+
export interface TrainingRuleBenchmarkRequest {
|
|
3834
|
+
benchmark_id: string | null;
|
|
3835
|
+
}
|
|
3617
3836
|
export interface TrainingRuleListResponse {
|
|
3618
3837
|
rules: TrainingRule[];
|
|
3619
3838
|
total: number;
|
|
@@ -3642,6 +3861,12 @@ export interface TrainingRunListResponse {
|
|
|
3642
3861
|
export interface TrainingRunResponse {
|
|
3643
3862
|
run: TrainingRun;
|
|
3644
3863
|
timeline: TrainingRunEvent[];
|
|
3864
|
+
/**
|
|
3865
|
+
* The standing benchmark's two absolute numbers for this run, beside the
|
|
3866
|
+
* paired verdict and never in place of it. Null when no benchmark is
|
|
3867
|
+
* attached to the rule.
|
|
3868
|
+
*/
|
|
3869
|
+
benchmark: BenchmarkRun | null;
|
|
3645
3870
|
links: TrainingRunLinks;
|
|
3646
3871
|
available_actions: TrainingRunActions;
|
|
3647
3872
|
}
|
|
@@ -3666,3 +3891,17 @@ export interface InferenceAliasResponse {
|
|
|
3666
3891
|
export interface InferenceAliasDeleteResponse {
|
|
3667
3892
|
deleted: boolean;
|
|
3668
3893
|
}
|
|
3894
|
+
export interface BenchmarkListResponse {
|
|
3895
|
+
benchmarks: Benchmark[];
|
|
3896
|
+
total: number;
|
|
3897
|
+
}
|
|
3898
|
+
export interface BenchmarkItemsResponse {
|
|
3899
|
+
items: BenchmarkItem[];
|
|
3900
|
+
total: number;
|
|
3901
|
+
}
|
|
3902
|
+
/** The trend, newest first. Every replay is a point, including the scoreless ones. */
|
|
3903
|
+
export interface BenchmarkHistoryResponse {
|
|
3904
|
+
benchmark_id: string;
|
|
3905
|
+
points: BenchmarkHistoryPoint[];
|
|
3906
|
+
total: number;
|
|
3907
|
+
}
|