runbios-sdk 0.2.1-dev.152 → 0.2.1-dev.158
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/dist/index.d.ts +2 -2
- package/dist/index.js +1 -1
- package/dist/resources/loop.d.ts +93 -1
- package/dist/resources/loop.js +133 -0
- package/dist/types.d.ts +261 -0
- package/package.json +1 -1
package/dist/index.d.ts
CHANGED
|
@@ -36,7 +36,7 @@ import { Loop } from './resources/loop.js';
|
|
|
36
36
|
* SDK version. Sent as part of the User-Agent header.
|
|
37
37
|
* Must match package.json "version" -- enforced by a contract test.
|
|
38
38
|
*/
|
|
39
|
-
export declare const VERSION = "0.2.1-dev.
|
|
39
|
+
export declare const VERSION = "0.2.1-dev.158";
|
|
40
40
|
export declare class RunBiOS {
|
|
41
41
|
/** Search models, fetch configs, check adapter compatibility. */
|
|
42
42
|
readonly models: Models;
|
|
@@ -75,6 +75,6 @@ export { Training } from './resources/training.js';
|
|
|
75
75
|
export { Wallet } from './resources/wallet.js';
|
|
76
76
|
export { GPU, type GPURecommendation } from './resources/gpu.js';
|
|
77
77
|
export { Inference, validateChatRequest, parseSSE, buildInferenceRequest, contextSizingBasis, type ContextSizingBasis, CONTEXT_DEFAULT_CEILING, CONTEXT_EDITABLE_FLOOR, type ChatMessage, type FunctionTool, type ChatCompletionParams, type ChatCompletionResponse, type ChatCompletionChunk, } from './resources/inference.js';
|
|
78
|
-
export type { BiOSConfig, PaginatedResponse, ApiErrorBody, AvailableGpuAlternative, CapacityMinimumRequirement, InferenceBookingAccepted, Model, ModelSearchParams, ModelSearchResponse, ModelDetailResponse, ModelConfig, Dataset, DatasetListParams, DatasetListResponse, DatasetUploadParams, DatasetPreview, DatasetPreviewParams, DatasetImportHFParams, DatasetRegisterHFParams, DatasetHubSearchParams, DatasetHubPreviewParams, DatasetValidation, DatasetFormatVariant, DatasetFormatSpec, DatasetFormatSpecs, DatasetStorageUsage, TrainingMethod, RLHFAlgorithm, AdapterType, TrainingJobStatus, TrainingStatusFilter, TrainingCreateParams, TrainingListParams, TrainingListResponse, TrainingJob, TrainingMetrics, TrainingResourceMetrics, TrainingDeviceMetrics, MetricPoint, MetricGraphConfig, TrainingCheckpoint, TrainingLogs, TrainingLogEntry, TrainingStopResponse, TrainingResumeResponse, CanonicalTrainingRequest, TrainingPreflightDataset, TrainingPreflightWarning, TrainingPreflightResponse, TrainingCapabilityChoice, TrainingConfigFieldCapability, TrainingCapabilities, GPUChoice, WalletBalance, Transaction, TransactionListResponse, TransactionListParams, GPUInfo, GPUPricingResponse, GPUOptionsParams, TrainingRecommendParams, TrainingAdvisorVerdict, TrainingAdvisorRecommendation, TrainingAdvisorUserShape, TrainingAdvisorJustification, TrainingAdvisorBasis, GPUOption, GPUOptionSuggestion, GPUOptionsResponse, InferenceStatus, InferenceCreateParams, InferenceUpdateParams, InferenceDeployment, InferenceDeploymentSummary, InferenceCreateResponse, InferenceListResponse, InferenceUpdateResponse, InferenceLifecycleResponse, InferenceDeleteResponse, InferenceGPUOptionMarket, InferenceGPUOption, InferenceGPUAlternative, InferenceGPUOptionsParams, InferenceGPUOptionsResponse, InferenceCanonicalRequest, InferencePreflightWarning, InferencePreflightResponse, AdapterCompatibility, AdapterCompatibilityResponse, AdapterCompatibilityParams, SupportedArchitecture, ArchitectureScope, SupportedArchitecturesResponse, ApiKey, ApiKeyScope, ApiKeyIntrospection, Organization, OrgMember, OrgInvite, Workspace, Integration, IntegrationCreateParams, StorageUsage, StorageObject, PromptTokensDetails, ChatCompletionUsage, TrainingCadence, TrainingCombinator, ServingKind, TrainingRulePausedReason, LoopTrainingMethod, TrainType, GradersScope, TrainingTrigger, TrainingRunState, TrainingVerdict, TrainingDecision, NotifyState, EvaluationStatus, EvaluationItemStatus, EvaluationWinner, EvaluationWarningCode, ConsentVia, TrainingToolCall, TrainingMessage, TrainingRuleServing, TrainingGPURung, TrainingRule, TrainingRuleConsent, TrainingRulePreflightRefusal, TrainingRuleKeyRef, TrainingRuleKeyCheck, TrainingRulePreflight, TrainingRuleBuildSpec, TrainingRuleTriggerInput, TrainingRuleTrainingInput, TrainingRuleDeployInput, TrainingRuleMoneyInput, TrainingRuleEvaluationInput, TrainingRulePromotionInput, TrainingRuleAcceptTerms, TrainingRulePreflightRequest, TrainingRuleCreateRequest, TrainingRuleUpdateRequest, TrainingRuleConsentRequest, TrainingRuleListParams, TrainingRunPromoteRequest, TrainingRunRejectRequest, TrainingRunRollbackRequest, TrainingRunCancelRequest, TrainingRunListParams, TrainingRun, TrainingRunSummary, TrainingRunEvent, TrainingRunLinks, TrainingRunActions, EvaluationRef, EvaluationDecoding, EvaluationJudgeDimension, EvaluationDimensionScore, EvaluationGraderScore, EvaluationWarning, EvaluationMargin, JudgeAgreementDimension, JudgeAgreement, Evaluation, EvaluationItemSide, EvaluationItem, EvaluationItemListParams, JudgeAgreementParams, AgentSettings, AgentSettingsRequest, InferenceAlias, InferenceAliasRequest, TrainingRuleListResponse, TrainingRuleResponse, TrainingRuleMutationResponse, TrainingRuleDeleteResponse, TrainingRunListResponse, TrainingRunResponse, TrainingRunActionResponse, EvaluationItemsResponse, AgentSettingsResponse, InferenceAliasListResponse, InferenceAliasResponse, InferenceAliasDeleteResponse, } from './types.js';
|
|
78
|
+
export type { BiOSConfig, PaginatedResponse, ApiErrorBody, AvailableGpuAlternative, CapacityMinimumRequirement, InferenceBookingAccepted, Model, ModelSearchParams, ModelSearchResponse, ModelDetailResponse, ModelConfig, Dataset, DatasetListParams, DatasetListResponse, DatasetUploadParams, DatasetPreview, DatasetPreviewParams, DatasetImportHFParams, DatasetRegisterHFParams, DatasetHubSearchParams, DatasetHubPreviewParams, DatasetValidation, DatasetFormatVariant, DatasetFormatSpec, DatasetFormatSpecs, DatasetStorageUsage, TrainingMethod, RLHFAlgorithm, AdapterType, TrainingJobStatus, TrainingStatusFilter, TrainingCreateParams, TrainingListParams, TrainingListResponse, TrainingJob, TrainingMetrics, TrainingResourceMetrics, TrainingDeviceMetrics, MetricPoint, MetricGraphConfig, TrainingCheckpoint, TrainingLogs, TrainingLogEntry, TrainingStopResponse, TrainingResumeResponse, CanonicalTrainingRequest, TrainingPreflightDataset, TrainingPreflightWarning, TrainingPreflightResponse, TrainingCapabilityChoice, TrainingConfigFieldCapability, TrainingCapabilities, GPUChoice, WalletBalance, Transaction, TransactionListResponse, TransactionListParams, GPUInfo, GPUPricingResponse, GPUOptionsParams, TrainingRecommendParams, TrainingAdvisorVerdict, TrainingAdvisorRecommendation, TrainingAdvisorUserShape, TrainingAdvisorJustification, TrainingAdvisorBasis, GPUOption, GPUOptionSuggestion, GPUOptionsResponse, InferenceStatus, InferenceCreateParams, InferenceUpdateParams, InferenceDeployment, InferenceDeploymentSummary, InferenceCreateResponse, InferenceListResponse, InferenceUpdateResponse, InferenceLifecycleResponse, InferenceDeleteResponse, InferenceGPUOptionMarket, InferenceGPUOption, InferenceGPUAlternative, InferenceGPUOptionsParams, InferenceGPUOptionsResponse, InferenceCanonicalRequest, InferencePreflightWarning, InferencePreflightResponse, AdapterCompatibility, AdapterCompatibilityResponse, AdapterCompatibilityParams, SupportedArchitecture, ArchitectureScope, SupportedArchitecturesResponse, ApiKey, ApiKeyScope, ApiKeyIntrospection, Organization, OrgMember, OrgInvite, Workspace, Integration, IntegrationCreateParams, StorageUsage, StorageObject, PromptTokensDetails, ChatCompletionUsage, TrainingCadence, TrainingCombinator, ServingKind, TrainingRulePausedReason, LoopTrainingMethod, TrainType, GradersScope, TrainingTrigger, TrainingRunState, TrainingVerdict, TrainingDecision, NotifyState, EvaluationStatus, EvaluationItemStatus, EvaluationWinner, EvaluationWarningCode, ConsentVia, TrainingToolCall, TrainingMessage, TrainingRuleServing, TrainingGPURung, TrainingRule, TrainingRuleConsent, TrainingRulePreflightRefusal, TrainingRuleKeyRef, TrainingRuleKeyCheck, TrainingRulePreflight, TrainingRuleBuildSpec, TrainingRuleTriggerInput, TrainingRuleTrainingInput, TrainingRuleDeployInput, TrainingRuleMoneyInput, TrainingRuleEvaluationInput, TrainingRulePromotionInput, TrainingRuleAcceptTerms, TrainingRulePreflightRequest, TrainingRuleCreateRequest, TrainingRuleUpdateRequest, TrainingRuleConsentRequest, TrainingRuleListParams, TrainingRunPromoteRequest, TrainingRunRejectRequest, TrainingRunRollbackRequest, TrainingRunCancelRequest, TrainingRunListParams, TrainingRun, TrainingRunSummary, TrainingRunEvent, TrainingRunLinks, TrainingRunActions, EvaluationRef, EvaluationDecoding, EvaluationJudgeDimension, EvaluationDimensionScore, EvaluationGraderScore, EvaluationWarning, EvaluationMargin, JudgeAgreementDimension, JudgeAgreement, Evaluation, EvaluationItemSide, EvaluationItem, EvaluationItemListParams, JudgeAgreementParams, AgentSettings, AgentSettingsRequest, BenchmarkStatus, BenchmarkSourceKind, BenchmarkRunStatus, BenchmarkScoring, BenchmarkDimensionScore, Benchmark, BenchmarkItem, BenchmarkRun, BenchmarkHistoryPoint, BenchmarkSource, BenchmarkCreateParams, BenchmarkListParams, BenchmarkItemListParams, BenchmarkHistoryParams, BenchmarkRetireParams, TrainingRuleBenchmarkRequest, InferenceAlias, InferenceAliasRequest, TrainingRuleListResponse, TrainingRuleResponse, TrainingRuleMutationResponse, TrainingRuleDeleteResponse, TrainingRunListResponse, TrainingRunResponse, TrainingRunActionResponse, EvaluationItemsResponse, AgentSettingsResponse, InferenceAliasListResponse, InferenceAliasResponse, InferenceAliasDeleteResponse, BenchmarkListResponse, BenchmarkItemsResponse, BenchmarkHistoryResponse, } from './types.js';
|
|
79
79
|
/** The closed set of run states a training run never leaves. */
|
|
80
80
|
export { TERMINAL_RUN_STATES } from './types.js';
|
package/dist/index.js
CHANGED
|
@@ -36,7 +36,7 @@ import { Loop } from './resources/loop.js';
|
|
|
36
36
|
* SDK version. Sent as part of the User-Agent header.
|
|
37
37
|
* Must match package.json "version" -- enforced by a contract test.
|
|
38
38
|
*/
|
|
39
|
-
export const VERSION = '0.2.1-dev.
|
|
39
|
+
export const VERSION = '0.2.1-dev.158';
|
|
40
40
|
export class RunBiOS {
|
|
41
41
|
/** Search models, fetch configs, check adapter compatibility. */
|
|
42
42
|
models;
|
package/dist/resources/loop.d.ts
CHANGED
|
@@ -1,5 +1,5 @@
|
|
|
1
1
|
import type { HttpClient } from '../client.js';
|
|
2
|
-
import type { LoopTrace, LoopTraceListResponse, LoopTraceListParams, LoopCaptureParams, LoopCaptureResult, LoopImportParams, LoopImportResult, LoopSignal, LoopSignalParams, LoopCandidate, LoopCandidateParams, LoopGrader, LoopGraderParams, LoopLabel, LoopLabelCount, LoopGradeResult, LoopDataset, LoopDatasetListParams, LoopDatasetItem, LoopDatasetCreateParams, LoopConfig, LoopConfigParams, LoopBuildRule, LoopBuildRuleParams, LoopBuildRuleUpdateParams, LoopStats, LoopJudge, LoopJudgeParams, LoopAgentCredential, LoopAgentStatus, LoopSampleRun, LoopSampleRunParams, LoopJudgeRun, LoopJudgeRunItems, LoopJudgeRunItemStatus, LoopJudgeWork, LoopJudgeVerdict, LoopJudgeVerdictResult, TrainingRule, TrainingRulePreflight, TrainingRulePreflightRequest, TrainingRuleCreateRequest, TrainingRuleUpdateRequest, TrainingRuleConsentRequest, TrainingRuleListParams, TrainingRuleListResponse, TrainingRuleResponse, TrainingRuleMutationResponse, TrainingRuleDeleteResponse, TrainingRun, TrainingRunListParams, TrainingRunListResponse, TrainingRunResponse, TrainingRunActionResponse, TrainingRunPromoteRequest, TrainingRunRejectRequest, TrainingRunRollbackRequest, TrainingRunCancelRequest, Evaluation, EvaluationItemListParams, EvaluationItemsResponse, JudgeAgreement, JudgeAgreementParams, AgentSettings, AgentSettingsRequest } from '../types.js';
|
|
2
|
+
import type { LoopTrace, LoopTraceListResponse, LoopTraceListParams, LoopCaptureParams, LoopCaptureResult, LoopImportParams, LoopImportResult, LoopSignal, LoopSignalParams, LoopCandidate, LoopCandidateParams, LoopGrader, LoopGraderParams, LoopLabel, LoopLabelCount, LoopGradeResult, LoopDataset, LoopDatasetListParams, LoopDatasetItem, LoopDatasetCreateParams, LoopConfig, LoopConfigParams, LoopBuildRule, LoopBuildRuleParams, LoopBuildRuleUpdateParams, LoopStats, LoopJudge, LoopJudgeParams, LoopAgentCredential, LoopAgentStatus, LoopSampleRun, LoopSampleRunParams, LoopJudgeRun, LoopJudgeRunItems, LoopJudgeRunItemStatus, LoopJudgeWork, LoopJudgeVerdict, LoopJudgeVerdictResult, TrainingRule, TrainingRulePreflight, TrainingRulePreflightRequest, TrainingRuleCreateRequest, TrainingRuleUpdateRequest, TrainingRuleConsentRequest, TrainingRuleListParams, TrainingRuleListResponse, TrainingRuleResponse, TrainingRuleMutationResponse, TrainingRuleDeleteResponse, TrainingRun, TrainingRunListParams, TrainingRunListResponse, TrainingRunResponse, TrainingRunActionResponse, TrainingRunPromoteRequest, TrainingRunRejectRequest, TrainingRunRollbackRequest, TrainingRunCancelRequest, Evaluation, EvaluationItemListParams, EvaluationItemsResponse, JudgeAgreement, JudgeAgreementParams, Benchmark, BenchmarkCreateParams, BenchmarkHistoryParams, BenchmarkHistoryResponse, BenchmarkItemListParams, BenchmarkItemsResponse, BenchmarkListParams, BenchmarkListResponse, BenchmarkRetireParams, BenchmarkRun, TrainingRuleBenchmarkRequest, AgentSettings, AgentSettingsRequest } from '../types.js';
|
|
3
3
|
/**
|
|
4
4
|
* The Conscious Loop -- capture what your model was asked and answered, record
|
|
5
5
|
* whether it was right, and turn those judgements into training data.
|
|
@@ -740,6 +740,98 @@ export declare class Loop {
|
|
|
740
740
|
* reviewer overlap yet" is an answer, and 100% of two pairs is not.
|
|
741
741
|
*/
|
|
742
742
|
getJudgeAgreement(judgeId: string, params?: JudgeAgreementParams): Promise<JudgeAgreement>;
|
|
743
|
+
/**
|
|
744
|
+
* Pin a fixed set of conversations, with a judge frozen beside them, and
|
|
745
|
+
* measure every future run against it.
|
|
746
|
+
*
|
|
747
|
+
* The comparison answers "is this candidate better than what serves today".
|
|
748
|
+
* It cannot answer "is my model getting better", because its rows, its
|
|
749
|
+
* opponent and its judge all move between runs. A benchmark is the other
|
|
750
|
+
* instrument: the same conversations, the same oracle, the same decoding,
|
|
751
|
+
* replayed against both models on every run that finishes its comparison,
|
|
752
|
+
* and reported as two absolute numbers on a scale that does not move.
|
|
753
|
+
*
|
|
754
|
+
* 10 to 200 conversations. A named conversation with nothing to ask a model
|
|
755
|
+
* is refused rather than skipped, because pinning 47 of the 50 you chose is
|
|
756
|
+
* the set being wrong from the first day and you would never find out.
|
|
757
|
+
*
|
|
758
|
+
* It raises no amount you have already agreed to. The replay's calls come
|
|
759
|
+
* out of the rule's existing `eval_ceiling_cents`, and
|
|
760
|
+
* `per_run_ceiling_cents` can only lower what is spent inside that.
|
|
761
|
+
*/
|
|
762
|
+
createBenchmark(params: BenchmarkCreateParams): Promise<Benchmark>;
|
|
763
|
+
/** Your benchmarks, newest first. Filter by `status` to hide retired ones. */
|
|
764
|
+
listBenchmarks(params?: BenchmarkListParams): Promise<BenchmarkListResponse>;
|
|
765
|
+
/**
|
|
766
|
+
* One benchmark: the frozen oracle, the scoring rule and the fingerprint of
|
|
767
|
+
* the set.
|
|
768
|
+
*
|
|
769
|
+
* `items_digest` is recomputed from the rows and compared at the start of
|
|
770
|
+
* every replay, so "the set cannot drift" is something you can check rather
|
|
771
|
+
* than something the platform promises.
|
|
772
|
+
*/
|
|
773
|
+
getBenchmark(id: string): Promise<Benchmark>;
|
|
774
|
+
/**
|
|
775
|
+
* The pinned conversations, paged. `limit` is capped at 100 by the service.
|
|
776
|
+
*
|
|
777
|
+
* `source_trace_id` and `source_trace_url` come back null once the
|
|
778
|
+
* conversation a row was copied from has been deleted. The row itself stays
|
|
779
|
+
* and every number already measured against it stays exactly as comparable
|
|
780
|
+
* as it was: retention cannot shrink a benchmark.
|
|
781
|
+
*/
|
|
782
|
+
listBenchmarkItems(id: string, params?: BenchmarkItemListParams): Promise<BenchmarkItemsResponse>;
|
|
783
|
+
/**
|
|
784
|
+
* The trend: every score this benchmark has produced, newest first, each
|
|
785
|
+
* beside the checkpoint that produced it and what you then did about it.
|
|
786
|
+
*
|
|
787
|
+
* EVERY REPLAY IS A POINT, including the ones with no numbers on them. A
|
|
788
|
+
* series that silently dropped the runs where the benchmark could not be
|
|
789
|
+
* scored would read as an unbroken line and not be one, so those points come
|
|
790
|
+
* back with `candidate_score` null and `status_reason` saying why.
|
|
791
|
+
*/
|
|
792
|
+
getBenchmarkHistory(id: string, params?: BenchmarkHistoryParams): Promise<BenchmarkHistoryResponse>;
|
|
793
|
+
/**
|
|
794
|
+
* Stop replaying a benchmark, and detach it from every rule that names it.
|
|
795
|
+
*
|
|
796
|
+
* Nothing measured is removed. There is no delete and no update on purpose:
|
|
797
|
+
* the series of numbers measured against a set is what a benchmark is for,
|
|
798
|
+
* so an edit would make everything before it incomparable with everything
|
|
799
|
+
* after it and a delete would throw the series away. Changing the set means
|
|
800
|
+
* pinning a new benchmark, and retiring the old one releases its name.
|
|
801
|
+
*
|
|
802
|
+
* Retiring twice is somebody pressing a button twice: the second call
|
|
803
|
+
* answers with the retired benchmark rather than a refusal.
|
|
804
|
+
*/
|
|
805
|
+
retireBenchmark(id: string, params?: BenchmarkRetireParams): Promise<Benchmark>;
|
|
806
|
+
/**
|
|
807
|
+
* One replay in full: both absolute scores, the difference between them on
|
|
808
|
+
* the same set, the per-dimension breakdown and what it cost.
|
|
809
|
+
*
|
|
810
|
+
* Read `status` before you read the scores. A replay that did not score
|
|
811
|
+
* every pinned conversation publishes no score at all -- all three of
|
|
812
|
+
* `candidate_score`, `incumbent_score` and `score_delta` are null and
|
|
813
|
+
* `status_reason` is the sentence that explains it -- because a mean over
|
|
814
|
+
* whichever conversations happened to succeed is a measurement of a
|
|
815
|
+
* different set, which is the exact defect a standing benchmark removes.
|
|
816
|
+
*/
|
|
817
|
+
getBenchmarkRun(id: string): Promise<BenchmarkRun>;
|
|
818
|
+
/**
|
|
819
|
+
* Attach a benchmark to a rule, or send `{ benchmark_id: null }` to detach
|
|
820
|
+
* it.
|
|
821
|
+
*
|
|
822
|
+
* Its own route rather than a field on the rule body, because it is a
|
|
823
|
+
* decision to replay a fixed set on every future run of this rule for as
|
|
824
|
+
* long as it stands, and its refusals -- retired, or belonging to another
|
|
825
|
+
* workspace -- are about the benchmark rather than about the rule.
|
|
826
|
+
*
|
|
827
|
+
* It does not invalidate consent and the reply says so: attaching raises
|
|
828
|
+
* neither the amount set aside for judge calls nor the amount set aside for
|
|
829
|
+
* keeping the new model available, so nobody is asked to read the same
|
|
830
|
+
* sentence again. A retired benchmark is refused with `BENCHMARK_RETIRED`,
|
|
831
|
+
* because a rule pointed at one would report no number on every run and say
|
|
832
|
+
* nothing about why.
|
|
833
|
+
*/
|
|
834
|
+
setTrainingRuleBenchmark(id: string, params: TrainingRuleBenchmarkRequest): Promise<TrainingRuleMutationResponse>;
|
|
743
835
|
/**
|
|
744
836
|
* The workspace's agent options: which model it defaults to, the system
|
|
745
837
|
* prompts it judges and samples with, and the monthly cap on what its model
|
package/dist/resources/loop.js
CHANGED
|
@@ -1031,6 +1031,139 @@ export class Loop {
|
|
|
1031
1031
|
const qs = q.toString();
|
|
1032
1032
|
return this._http.fetchGet(`/api/loop/judges/${encodeURIComponent(judgeId)}/agreement${qs ? `?${qs}` : ''}`);
|
|
1033
1033
|
}
|
|
1034
|
+
// ── the standing benchmark ────────────────────────────────────────────
|
|
1035
|
+
/**
|
|
1036
|
+
* Pin a fixed set of conversations, with a judge frozen beside them, and
|
|
1037
|
+
* measure every future run against it.
|
|
1038
|
+
*
|
|
1039
|
+
* The comparison answers "is this candidate better than what serves today".
|
|
1040
|
+
* It cannot answer "is my model getting better", because its rows, its
|
|
1041
|
+
* opponent and its judge all move between runs. A benchmark is the other
|
|
1042
|
+
* instrument: the same conversations, the same oracle, the same decoding,
|
|
1043
|
+
* replayed against both models on every run that finishes its comparison,
|
|
1044
|
+
* and reported as two absolute numbers on a scale that does not move.
|
|
1045
|
+
*
|
|
1046
|
+
* 10 to 200 conversations. A named conversation with nothing to ask a model
|
|
1047
|
+
* is refused rather than skipped, because pinning 47 of the 50 you chose is
|
|
1048
|
+
* the set being wrong from the first day and you would never find out.
|
|
1049
|
+
*
|
|
1050
|
+
* It raises no amount you have already agreed to. The replay's calls come
|
|
1051
|
+
* out of the rule's existing `eval_ceiling_cents`, and
|
|
1052
|
+
* `per_run_ceiling_cents` can only lower what is spent inside that.
|
|
1053
|
+
*/
|
|
1054
|
+
async createBenchmark(params) {
|
|
1055
|
+
const res = await this._http.fetchPost('/api/loop/benchmarks', params);
|
|
1056
|
+
return res.benchmark;
|
|
1057
|
+
}
|
|
1058
|
+
/** Your benchmarks, newest first. Filter by `status` to hide retired ones. */
|
|
1059
|
+
async listBenchmarks(params = {}) {
|
|
1060
|
+
const q = new URLSearchParams();
|
|
1061
|
+
if (params.status)
|
|
1062
|
+
q.set('status', params.status);
|
|
1063
|
+
if (params.limit != null)
|
|
1064
|
+
q.set('limit', String(params.limit));
|
|
1065
|
+
if (params.offset != null)
|
|
1066
|
+
q.set('offset', String(params.offset));
|
|
1067
|
+
const qs = q.toString();
|
|
1068
|
+
return this._http.fetchGet(`/api/loop/benchmarks${qs ? `?${qs}` : ''}`);
|
|
1069
|
+
}
|
|
1070
|
+
/**
|
|
1071
|
+
* One benchmark: the frozen oracle, the scoring rule and the fingerprint of
|
|
1072
|
+
* the set.
|
|
1073
|
+
*
|
|
1074
|
+
* `items_digest` is recomputed from the rows and compared at the start of
|
|
1075
|
+
* every replay, so "the set cannot drift" is something you can check rather
|
|
1076
|
+
* than something the platform promises.
|
|
1077
|
+
*/
|
|
1078
|
+
async getBenchmark(id) {
|
|
1079
|
+
const res = await this._http.fetchGet(`/api/loop/benchmarks/${encodeURIComponent(id)}`);
|
|
1080
|
+
return res.benchmark;
|
|
1081
|
+
}
|
|
1082
|
+
/**
|
|
1083
|
+
* The pinned conversations, paged. `limit` is capped at 100 by the service.
|
|
1084
|
+
*
|
|
1085
|
+
* `source_trace_id` and `source_trace_url` come back null once the
|
|
1086
|
+
* conversation a row was copied from has been deleted. The row itself stays
|
|
1087
|
+
* and every number already measured against it stays exactly as comparable
|
|
1088
|
+
* as it was: retention cannot shrink a benchmark.
|
|
1089
|
+
*/
|
|
1090
|
+
async listBenchmarkItems(id, params = {}) {
|
|
1091
|
+
const q = new URLSearchParams();
|
|
1092
|
+
if (params.limit != null)
|
|
1093
|
+
q.set('limit', String(params.limit));
|
|
1094
|
+
if (params.offset != null)
|
|
1095
|
+
q.set('offset', String(params.offset));
|
|
1096
|
+
const qs = q.toString();
|
|
1097
|
+
return this._http.fetchGet(`/api/loop/benchmarks/${encodeURIComponent(id)}/items${qs ? `?${qs}` : ''}`);
|
|
1098
|
+
}
|
|
1099
|
+
/**
|
|
1100
|
+
* The trend: every score this benchmark has produced, newest first, each
|
|
1101
|
+
* beside the checkpoint that produced it and what you then did about it.
|
|
1102
|
+
*
|
|
1103
|
+
* EVERY REPLAY IS A POINT, including the ones with no numbers on them. A
|
|
1104
|
+
* series that silently dropped the runs where the benchmark could not be
|
|
1105
|
+
* scored would read as an unbroken line and not be one, so those points come
|
|
1106
|
+
* back with `candidate_score` null and `status_reason` saying why.
|
|
1107
|
+
*/
|
|
1108
|
+
async getBenchmarkHistory(id, params = {}) {
|
|
1109
|
+
const q = new URLSearchParams();
|
|
1110
|
+
if (params.limit != null)
|
|
1111
|
+
q.set('limit', String(params.limit));
|
|
1112
|
+
if (params.offset != null)
|
|
1113
|
+
q.set('offset', String(params.offset));
|
|
1114
|
+
const qs = q.toString();
|
|
1115
|
+
return this._http.fetchGet(`/api/loop/benchmarks/${encodeURIComponent(id)}/history${qs ? `?${qs}` : ''}`);
|
|
1116
|
+
}
|
|
1117
|
+
/**
|
|
1118
|
+
* Stop replaying a benchmark, and detach it from every rule that names it.
|
|
1119
|
+
*
|
|
1120
|
+
* Nothing measured is removed. There is no delete and no update on purpose:
|
|
1121
|
+
* the series of numbers measured against a set is what a benchmark is for,
|
|
1122
|
+
* so an edit would make everything before it incomparable with everything
|
|
1123
|
+
* after it and a delete would throw the series away. Changing the set means
|
|
1124
|
+
* pinning a new benchmark, and retiring the old one releases its name.
|
|
1125
|
+
*
|
|
1126
|
+
* Retiring twice is somebody pressing a button twice: the second call
|
|
1127
|
+
* answers with the retired benchmark rather than a refusal.
|
|
1128
|
+
*/
|
|
1129
|
+
async retireBenchmark(id, params = {}) {
|
|
1130
|
+
const res = await this._http.fetchPost(`/api/loop/benchmarks/${encodeURIComponent(id)}/retire`, params);
|
|
1131
|
+
return res.benchmark;
|
|
1132
|
+
}
|
|
1133
|
+
/**
|
|
1134
|
+
* One replay in full: both absolute scores, the difference between them on
|
|
1135
|
+
* the same set, the per-dimension breakdown and what it cost.
|
|
1136
|
+
*
|
|
1137
|
+
* Read `status` before you read the scores. A replay that did not score
|
|
1138
|
+
* every pinned conversation publishes no score at all -- all three of
|
|
1139
|
+
* `candidate_score`, `incumbent_score` and `score_delta` are null and
|
|
1140
|
+
* `status_reason` is the sentence that explains it -- because a mean over
|
|
1141
|
+
* whichever conversations happened to succeed is a measurement of a
|
|
1142
|
+
* different set, which is the exact defect a standing benchmark removes.
|
|
1143
|
+
*/
|
|
1144
|
+
async getBenchmarkRun(id) {
|
|
1145
|
+
const res = await this._http.fetchGet(`/api/loop/benchmark-runs/${encodeURIComponent(id)}`);
|
|
1146
|
+
return res.benchmark_run;
|
|
1147
|
+
}
|
|
1148
|
+
/**
|
|
1149
|
+
* Attach a benchmark to a rule, or send `{ benchmark_id: null }` to detach
|
|
1150
|
+
* it.
|
|
1151
|
+
*
|
|
1152
|
+
* Its own route rather than a field on the rule body, because it is a
|
|
1153
|
+
* decision to replay a fixed set on every future run of this rule for as
|
|
1154
|
+
* long as it stands, and its refusals -- retired, or belonging to another
|
|
1155
|
+
* workspace -- are about the benchmark rather than about the rule.
|
|
1156
|
+
*
|
|
1157
|
+
* It does not invalidate consent and the reply says so: attaching raises
|
|
1158
|
+
* neither the amount set aside for judge calls nor the amount set aside for
|
|
1159
|
+
* keeping the new model available, so nobody is asked to read the same
|
|
1160
|
+
* sentence again. A retired benchmark is refused with `BENCHMARK_RETIRED`,
|
|
1161
|
+
* because a rule pointed at one would report no number on every run and say
|
|
1162
|
+
* nothing about why.
|
|
1163
|
+
*/
|
|
1164
|
+
async setTrainingRuleBenchmark(id, params) {
|
|
1165
|
+
return this._http.fetchPut(`/api/loop/training-rules/${encodeURIComponent(id)}/benchmark`, params);
|
|
1166
|
+
}
|
|
1034
1167
|
// ── agent settings ────────────────────────────────────────────────────
|
|
1035
1168
|
/**
|
|
1036
1169
|
* The workspace's agent options: which model it defaults to, the system
|
package/dist/types.d.ts
CHANGED
|
@@ -3083,6 +3083,15 @@ export interface TrainingRule {
|
|
|
3083
3083
|
eval_max_rows: number;
|
|
3084
3084
|
eval_max_tokens: number;
|
|
3085
3085
|
min_holdout_rows: number;
|
|
3086
|
+
/**
|
|
3087
|
+
* The standing benchmark replayed on every run of this rule, beside the
|
|
3088
|
+
* per-run comparison and never instead of it. Null is the ordinary state.
|
|
3089
|
+
*
|
|
3090
|
+
* Set with `setTrainingRuleBenchmark`, not by an update to the rule: it is a
|
|
3091
|
+
* decision to replay a fixed set on every future run, and it has refusals of
|
|
3092
|
+
* its own. It raises no amount, so it does not invalidate consent.
|
|
3093
|
+
*/
|
|
3094
|
+
benchmark_id: string | null;
|
|
3086
3095
|
auto_promote: boolean;
|
|
3087
3096
|
promote_margin: number;
|
|
3088
3097
|
promote_min_win_rate: number;
|
|
@@ -3275,6 +3284,21 @@ export interface TrainingRulePreflightRequest {
|
|
|
3275
3284
|
export interface TrainingRuleCreateRequest extends TrainingRulePreflightRequest {
|
|
3276
3285
|
/** Absent is a refusal, not a default. */
|
|
3277
3286
|
accept_terms?: TrainingRuleAcceptTerms;
|
|
3287
|
+
/**
|
|
3288
|
+
* The standing benchmark this rule replays, set here so it applies to the
|
|
3289
|
+
* rule's FIRST run.
|
|
3290
|
+
*
|
|
3291
|
+
* A rule can fire within seconds of being created, and every step of a run
|
|
3292
|
+
* reads the benchmark off the rule as it stood at that moment. Attaching one
|
|
3293
|
+
* afterwards with `setTrainingRuleBenchmark` applies from the next run and
|
|
3294
|
+
* says nothing about the first, which is the run somebody is watching.
|
|
3295
|
+
*
|
|
3296
|
+
* Refused on the same terms the attach route refuses it: `404` when it is not
|
|
3297
|
+
* this workspace's, `409 BENCHMARK_RETIRED` when it has been retired. Use
|
|
3298
|
+
* `setTrainingRuleBenchmark` to change or remove it later; it is not on the
|
|
3299
|
+
* preflight body and not on the update body.
|
|
3300
|
+
*/
|
|
3301
|
+
benchmark_id?: string;
|
|
3278
3302
|
}
|
|
3279
3303
|
export interface TrainingRuleUpdateRequest extends TrainingRulePreflightRequest {
|
|
3280
3304
|
expected_revision?: number;
|
|
@@ -3352,6 +3376,15 @@ export interface TrainingRun {
|
|
|
3352
3376
|
candidate_deadline_at: string | null;
|
|
3353
3377
|
candidate_cleanup_at: string | null;
|
|
3354
3378
|
evaluation_id: string | null;
|
|
3379
|
+
/** This run's replay of the rule's standing benchmark, if it had one. */
|
|
3380
|
+
benchmark_run_id: string | null;
|
|
3381
|
+
/**
|
|
3382
|
+
* What the replay's calls cost. Kept apart from `spent_eval_cents` so a
|
|
3383
|
+
* report can say what the comparison cost and what the benchmark cost; both
|
|
3384
|
+
* come out of `eval_ceiling_cents` and their sum can never exceed it, so
|
|
3385
|
+
* anything totalling what a run cost has to add this one too.
|
|
3386
|
+
*/
|
|
3387
|
+
benchmark_spent_cents: number;
|
|
3355
3388
|
alias_name: string | null;
|
|
3356
3389
|
alias_written_at: string | null;
|
|
3357
3390
|
serving_before: TrainingRuleServing | null;
|
|
@@ -3389,6 +3422,8 @@ export interface TrainingRunSummary {
|
|
|
3389
3422
|
billed_training_cents: number;
|
|
3390
3423
|
billed_candidate_cents: number;
|
|
3391
3424
|
spent_eval_cents: number;
|
|
3425
|
+
/** The fourth money column. A row that leaves it out adds up short. */
|
|
3426
|
+
benchmark_spent_cents: number;
|
|
3392
3427
|
training_job_id: string | null;
|
|
3393
3428
|
candidate_deployment_id: string | null;
|
|
3394
3429
|
evaluation_id: string | null;
|
|
@@ -3414,6 +3449,8 @@ export interface TrainingRunLinks {
|
|
|
3414
3449
|
candidate_url: string | null;
|
|
3415
3450
|
dataset_url: string | null;
|
|
3416
3451
|
evaluation_url: string | null;
|
|
3452
|
+
/** The TREND the benchmark number belongs to, not the one replay. */
|
|
3453
|
+
benchmark_url: string | null;
|
|
3417
3454
|
}
|
|
3418
3455
|
/** What this reader may do right now. A button that cannot work is never shown. */
|
|
3419
3456
|
export interface TrainingRunActions {
|
|
@@ -3614,6 +3651,203 @@ export interface InferenceAliasRequest {
|
|
|
3614
3651
|
target_inference_id: string;
|
|
3615
3652
|
origin?: string;
|
|
3616
3653
|
}
|
|
3654
|
+
/**
|
|
3655
|
+
* A benchmark is active until it is retired. There is no delete and no update:
|
|
3656
|
+
* the series of numbers measured against a set is what a benchmark is for, so
|
|
3657
|
+
* an edit would make every number before it incomparable with every number
|
|
3658
|
+
* after it, and a delete throws the series away.
|
|
3659
|
+
*/
|
|
3660
|
+
export type BenchmarkStatus = 'active' | 'retired';
|
|
3661
|
+
/** Where the pinned conversations are copied from. Read once, at creation. */
|
|
3662
|
+
export type BenchmarkSourceKind = 'traces' | 'dataset';
|
|
3663
|
+
/**
|
|
3664
|
+
* The life of one replay.
|
|
3665
|
+
*
|
|
3666
|
+
* `budget_stopped` reached the amount left for it inside the run's own ceiling,
|
|
3667
|
+
* and `abandoned` ran out of the time the run sets aside for it. Both are
|
|
3668
|
+
* points on the trend carrying `status_reason`, never silent gaps.
|
|
3669
|
+
*/
|
|
3670
|
+
export type BenchmarkRunStatus = 'open' | 'done' | 'failed' | 'budget_stopped' | 'abandoned';
|
|
3671
|
+
/**
|
|
3672
|
+
* How the pinned conversations become one number, frozen on the benchmark.
|
|
3673
|
+
*
|
|
3674
|
+
* `require_all_rows` is what makes the number comparable at all: a mean over
|
|
3675
|
+
* whichever conversations happened to succeed is a measurement of a different
|
|
3676
|
+
* set, so a short replay publishes no score and says why. A partial benchmark
|
|
3677
|
+
* is a missing number, never a lower one.
|
|
3678
|
+
*/
|
|
3679
|
+
export interface BenchmarkScoring {
|
|
3680
|
+
metric: string;
|
|
3681
|
+
scale: string;
|
|
3682
|
+
aggregate: string;
|
|
3683
|
+
require_all_rows: boolean;
|
|
3684
|
+
}
|
|
3685
|
+
/**
|
|
3686
|
+
* One rubric dimension of one replay, for both models.
|
|
3687
|
+
*
|
|
3688
|
+
* Deliberately not `EvaluationDimensionScore`: these are absolute means on a
|
|
3689
|
+
* fixed set and those are paired means on that run's own held-back rows.
|
|
3690
|
+
*/
|
|
3691
|
+
export interface BenchmarkDimensionScore {
|
|
3692
|
+
dimension: string;
|
|
3693
|
+
incumbent: number;
|
|
3694
|
+
candidate: number;
|
|
3695
|
+
delta: number;
|
|
3696
|
+
}
|
|
3697
|
+
/**
|
|
3698
|
+
* The frozen definition: the oracle, the scoring rule and the fingerprint of
|
|
3699
|
+
* the set. The conversations themselves are their own paged route.
|
|
3700
|
+
*/
|
|
3701
|
+
export interface Benchmark {
|
|
3702
|
+
id: string;
|
|
3703
|
+
workspace_id: string;
|
|
3704
|
+
name: string;
|
|
3705
|
+
status: BenchmarkStatus;
|
|
3706
|
+
/** Provenance only: everything needed from the judge is copied below it. */
|
|
3707
|
+
judge_id: string | null;
|
|
3708
|
+
judge_name: string;
|
|
3709
|
+
judge_model: string;
|
|
3710
|
+
judge_instructions: string;
|
|
3711
|
+
judge_dimensions: EvaluationJudgeDimension[];
|
|
3712
|
+
judge_system_prompt: string | null;
|
|
3713
|
+
decoding: EvaluationDecoding;
|
|
3714
|
+
scoring: BenchmarkScoring;
|
|
3715
|
+
item_count: number;
|
|
3716
|
+
/** Recomputed from the rows and compared at the start of every replay. */
|
|
3717
|
+
items_digest: string;
|
|
3718
|
+
/** A cap INSIDE the run's own amount for judge calls, never an addition to it. */
|
|
3719
|
+
per_run_ceiling_cents: number;
|
|
3720
|
+
created_by: string;
|
|
3721
|
+
created_at: string;
|
|
3722
|
+
retired_at: string | null;
|
|
3723
|
+
retired_by: string | null;
|
|
3724
|
+
}
|
|
3725
|
+
/**
|
|
3726
|
+
* One pinned conversation.
|
|
3727
|
+
*
|
|
3728
|
+
* `source_trace_id` is a link and is allowed to go null: the prompt was copied
|
|
3729
|
+
* at pin time, so retention removing the conversation takes away the ability to
|
|
3730
|
+
* open it and takes away nothing else. Every number already measured stays
|
|
3731
|
+
* exactly as comparable as it was.
|
|
3732
|
+
*/
|
|
3733
|
+
export interface BenchmarkItem {
|
|
3734
|
+
id: number;
|
|
3735
|
+
ordinal: number;
|
|
3736
|
+
source_trace_id: string | null;
|
|
3737
|
+
source_trace_url: string | null;
|
|
3738
|
+
prompt: TrainingMessage[];
|
|
3739
|
+
tools: unknown[] | null;
|
|
3740
|
+
}
|
|
3741
|
+
/**
|
|
3742
|
+
* One replay: this benchmark, on this training run, against both models.
|
|
3743
|
+
*
|
|
3744
|
+
* IT DOES NOT GATE PROMOTION. The paired comparison applies the judge to both
|
|
3745
|
+
* sides of one conversation in one pass, so most of the judge's variance
|
|
3746
|
+
* cancels in its delta; an absolute mean carries that variance whole. What
|
|
3747
|
+
* ships here is the number, the difference against what serves today on the
|
|
3748
|
+
* same set, and the history. A person reads the trend.
|
|
3749
|
+
*/
|
|
3750
|
+
export interface BenchmarkRun {
|
|
3751
|
+
id: string;
|
|
3752
|
+
benchmark_id: string;
|
|
3753
|
+
benchmark_name: string;
|
|
3754
|
+
run_id: string;
|
|
3755
|
+
workspace_id: string;
|
|
3756
|
+
incumbent_ref: EvaluationRef;
|
|
3757
|
+
candidate_ref: EvaluationRef;
|
|
3758
|
+
/** The set actually scored, recorded beside the score. */
|
|
3759
|
+
items_digest: string;
|
|
3760
|
+
rows_total: number;
|
|
3761
|
+
rows_scored: number;
|
|
3762
|
+
rows_failed: number;
|
|
3763
|
+
/** Absolute, on the frozen judge's scale, and null unless every row scored. */
|
|
3764
|
+
candidate_score: number | null;
|
|
3765
|
+
incumbent_score: number | null;
|
|
3766
|
+
score_delta: number | null;
|
|
3767
|
+
per_dimension: BenchmarkDimensionScore[];
|
|
3768
|
+
prompt_tokens: number;
|
|
3769
|
+
completion_tokens: number;
|
|
3770
|
+
spent_cents: number;
|
|
3771
|
+
ceiling_cents: number;
|
|
3772
|
+
status: BenchmarkRunStatus;
|
|
3773
|
+
/** The whole explanation when there is no score, so never empty on one. */
|
|
3774
|
+
status_reason: string | null;
|
|
3775
|
+
created_at: string;
|
|
3776
|
+
finished_at: string | null;
|
|
3777
|
+
}
|
|
3778
|
+
/**
|
|
3779
|
+
* One point on the trend line.
|
|
3780
|
+
*
|
|
3781
|
+
* It carries the checkpoint and what the person then decided, because a trend
|
|
3782
|
+
* with no idea what changed between two points is a chart rather than an
|
|
3783
|
+
* answer.
|
|
3784
|
+
*/
|
|
3785
|
+
export interface BenchmarkHistoryPoint {
|
|
3786
|
+
benchmark_run_id: string;
|
|
3787
|
+
run_id: string;
|
|
3788
|
+
run_seq: number;
|
|
3789
|
+
rule_id: string | null;
|
|
3790
|
+
rule_name: string | null;
|
|
3791
|
+
created_at: string;
|
|
3792
|
+
candidate_score: number | null;
|
|
3793
|
+
incumbent_score: number | null;
|
|
3794
|
+
score_delta: number | null;
|
|
3795
|
+
checkpoint_id: string | null;
|
|
3796
|
+
decision: TrainingDecision | null;
|
|
3797
|
+
status: BenchmarkRunStatus;
|
|
3798
|
+
status_reason: string | null;
|
|
3799
|
+
}
|
|
3800
|
+
/** Where the conversations are copied FROM. Read once, at creation, never again. */
|
|
3801
|
+
export interface BenchmarkSource {
|
|
3802
|
+
kind: BenchmarkSourceKind;
|
|
3803
|
+
trace_ids?: string[];
|
|
3804
|
+
dataset_id?: string;
|
|
3805
|
+
split?: string;
|
|
3806
|
+
}
|
|
3807
|
+
/**
|
|
3808
|
+
* The pin: 10 to 200 conversations, and a judge with a rubric to measure them.
|
|
3809
|
+
*
|
|
3810
|
+
* A conversation with nothing to ask a model is refused rather than skipped.
|
|
3811
|
+
* Pinning 47 of the 50 somebody chose is the set being wrong from the first
|
|
3812
|
+
* day, and they would never find out.
|
|
3813
|
+
*/
|
|
3814
|
+
export interface BenchmarkCreateParams {
|
|
3815
|
+
name: string;
|
|
3816
|
+
judge_id: string;
|
|
3817
|
+
source: BenchmarkSource;
|
|
3818
|
+
/** Absent uses the judge's own model, and absent that the workspace default. */
|
|
3819
|
+
judge_model?: string;
|
|
3820
|
+
/** What both models are given to answer in. A shorter answer is a different answer. */
|
|
3821
|
+
max_tokens?: number;
|
|
3822
|
+
per_run_ceiling_cents: number;
|
|
3823
|
+
}
|
|
3824
|
+
export interface BenchmarkListParams {
|
|
3825
|
+
status?: BenchmarkStatus;
|
|
3826
|
+
limit?: number;
|
|
3827
|
+
offset?: number;
|
|
3828
|
+
}
|
|
3829
|
+
export interface BenchmarkItemListParams {
|
|
3830
|
+
/** Capped at 100 by the service. */
|
|
3831
|
+
limit?: number;
|
|
3832
|
+
offset?: number;
|
|
3833
|
+
}
|
|
3834
|
+
export interface BenchmarkHistoryParams {
|
|
3835
|
+
limit?: number;
|
|
3836
|
+
offset?: number;
|
|
3837
|
+
}
|
|
3838
|
+
/** A reason is a courtesy here, not a requirement. */
|
|
3839
|
+
export interface BenchmarkRetireParams {
|
|
3840
|
+
reason?: string;
|
|
3841
|
+
}
|
|
3842
|
+
/**
|
|
3843
|
+
* Attach a benchmark to a rule, or detach it with a present null.
|
|
3844
|
+
*
|
|
3845
|
+
* Not optional, and not omittable: this route sets the field, so an absent key
|
|
3846
|
+
* would be a request with nothing in it. Send the id to attach, null to detach.
|
|
3847
|
+
*/
|
|
3848
|
+
export interface TrainingRuleBenchmarkRequest {
|
|
3849
|
+
benchmark_id: string | null;
|
|
3850
|
+
}
|
|
3617
3851
|
export interface TrainingRuleListResponse {
|
|
3618
3852
|
rules: TrainingRule[];
|
|
3619
3853
|
total: number;
|
|
@@ -3633,6 +3867,13 @@ export interface TrainingRuleMutationResponse {
|
|
|
3633
3867
|
export interface TrainingRuleDeleteResponse {
|
|
3634
3868
|
deleted: boolean;
|
|
3635
3869
|
rule_id: string;
|
|
3870
|
+
/**
|
|
3871
|
+
* The name this delete just spent. The delete is a soft delete and the name
|
|
3872
|
+
* index carries no partial predicate, so the row goes on holding the name
|
|
3873
|
+
* after it has left every list you can read, and no later rule in the
|
|
3874
|
+
* workspace can be called that. There is no purge and no undelete.
|
|
3875
|
+
*/
|
|
3876
|
+
retained_name: string;
|
|
3636
3877
|
cancelled_run_id: string | null;
|
|
3637
3878
|
}
|
|
3638
3879
|
export interface TrainingRunListResponse {
|
|
@@ -3642,6 +3883,12 @@ export interface TrainingRunListResponse {
|
|
|
3642
3883
|
export interface TrainingRunResponse {
|
|
3643
3884
|
run: TrainingRun;
|
|
3644
3885
|
timeline: TrainingRunEvent[];
|
|
3886
|
+
/**
|
|
3887
|
+
* The standing benchmark's two absolute numbers for this run, beside the
|
|
3888
|
+
* paired verdict and never in place of it. Null when no benchmark is
|
|
3889
|
+
* attached to the rule.
|
|
3890
|
+
*/
|
|
3891
|
+
benchmark: BenchmarkRun | null;
|
|
3645
3892
|
links: TrainingRunLinks;
|
|
3646
3893
|
available_actions: TrainingRunActions;
|
|
3647
3894
|
}
|
|
@@ -3666,3 +3913,17 @@ export interface InferenceAliasResponse {
|
|
|
3666
3913
|
export interface InferenceAliasDeleteResponse {
|
|
3667
3914
|
deleted: boolean;
|
|
3668
3915
|
}
|
|
3916
|
+
export interface BenchmarkListResponse {
|
|
3917
|
+
benchmarks: Benchmark[];
|
|
3918
|
+
total: number;
|
|
3919
|
+
}
|
|
3920
|
+
export interface BenchmarkItemsResponse {
|
|
3921
|
+
items: BenchmarkItem[];
|
|
3922
|
+
total: number;
|
|
3923
|
+
}
|
|
3924
|
+
/** The trend, newest first. Every replay is a point, including the scoreless ones. */
|
|
3925
|
+
export interface BenchmarkHistoryResponse {
|
|
3926
|
+
benchmark_id: string;
|
|
3927
|
+
points: BenchmarkHistoryPoint[];
|
|
3928
|
+
total: number;
|
|
3929
|
+
}
|