runbios-sdk 0.2.1-dev.137 → 0.2.1-dev.140
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/dist/index.d.ts +4 -2
- package/dist/index.js +3 -1
- package/dist/resources/inference.d.ts +26 -1
- package/dist/resources/inference.js +45 -0
- package/dist/resources/loop.d.ts +196 -1
- package/dist/resources/loop.js +302 -0
- package/dist/types.d.ts +685 -0
- package/dist/types.js +11 -1
- package/package.json +1 -1
package/dist/index.d.ts
CHANGED
|
@@ -36,7 +36,7 @@ import { Loop } from './resources/loop.js';
|
|
|
36
36
|
* SDK version. Sent as part of the User-Agent header.
|
|
37
37
|
* Must match package.json "version" -- enforced by a contract test.
|
|
38
38
|
*/
|
|
39
|
-
export declare const VERSION = "0.2.1-dev.
|
|
39
|
+
export declare const VERSION = "0.2.1-dev.140";
|
|
40
40
|
export declare class RunBiOS {
|
|
41
41
|
/** Search models, fetch configs, check adapter compatibility. */
|
|
42
42
|
readonly models: Models;
|
|
@@ -75,4 +75,6 @@ export { Training } from './resources/training.js';
|
|
|
75
75
|
export { Wallet } from './resources/wallet.js';
|
|
76
76
|
export { GPU, type GPURecommendation } from './resources/gpu.js';
|
|
77
77
|
export { Inference, validateChatRequest, parseSSE, buildInferenceRequest, contextSizingBasis, type ContextSizingBasis, CONTEXT_DEFAULT_CEILING, CONTEXT_EDITABLE_FLOOR, type ChatMessage, type FunctionTool, type ChatCompletionParams, type ChatCompletionResponse, type ChatCompletionChunk, } from './resources/inference.js';
|
|
78
|
-
export type { BiOSConfig, PaginatedResponse, ApiErrorBody, AvailableGpuAlternative, CapacityMinimumRequirement, InferenceBookingAccepted, Model, ModelSearchParams, ModelSearchResponse, ModelDetailResponse, ModelConfig, Dataset, DatasetListParams, DatasetListResponse, DatasetUploadParams, DatasetPreview, DatasetPreviewParams, DatasetImportHFParams, DatasetRegisterHFParams, DatasetHubSearchParams, DatasetHubPreviewParams, DatasetValidation, DatasetFormatVariant, DatasetFormatSpec, DatasetFormatSpecs, DatasetStorageUsage, TrainingMethod, RLHFAlgorithm, AdapterType, TrainingJobStatus, TrainingStatusFilter, TrainingCreateParams, TrainingListParams, TrainingListResponse, TrainingJob, TrainingMetrics, TrainingResourceMetrics, TrainingDeviceMetrics, MetricPoint, MetricGraphConfig, TrainingCheckpoint, TrainingLogs, TrainingLogEntry, TrainingStopResponse, TrainingResumeResponse, CanonicalTrainingRequest, TrainingPreflightDataset, TrainingPreflightWarning, TrainingPreflightResponse, TrainingCapabilityChoice, TrainingConfigFieldCapability, TrainingCapabilities, GPUChoice, WalletBalance, Transaction, TransactionListResponse, TransactionListParams, GPUInfo, GPUPricingResponse, GPUOptionsParams, TrainingRecommendParams, TrainingAdvisorVerdict, TrainingAdvisorRecommendation, TrainingAdvisorUserShape, TrainingAdvisorJustification, TrainingAdvisorBasis, GPUOption, GPUOptionSuggestion, GPUOptionsResponse, InferenceStatus, InferenceCreateParams, InferenceUpdateParams, InferenceDeployment, InferenceDeploymentSummary, InferenceCreateResponse, InferenceListResponse, InferenceUpdateResponse, InferenceLifecycleResponse, InferenceDeleteResponse, InferenceGPUOptionMarket, InferenceGPUOption, InferenceGPUAlternative, InferenceGPUOptionsParams, InferenceGPUOptionsResponse, InferenceCanonicalRequest, InferencePreflightWarning, InferencePreflightResponse, AdapterCompatibility, AdapterCompatibilityResponse, AdapterCompatibilityParams, SupportedArchitecture, ArchitectureScope, SupportedArchitecturesResponse, ApiKey, ApiKeyScope, ApiKeyIntrospection, Organization, OrgMember, OrgInvite, Workspace, Integration, IntegrationCreateParams, StorageUsage, StorageObject, PromptTokensDetails, ChatCompletionUsage, } from './types.js';
|
|
78
|
+
export type { BiOSConfig, PaginatedResponse, ApiErrorBody, AvailableGpuAlternative, CapacityMinimumRequirement, InferenceBookingAccepted, Model, ModelSearchParams, ModelSearchResponse, ModelDetailResponse, ModelConfig, Dataset, DatasetListParams, DatasetListResponse, DatasetUploadParams, DatasetPreview, DatasetPreviewParams, DatasetImportHFParams, DatasetRegisterHFParams, DatasetHubSearchParams, DatasetHubPreviewParams, DatasetValidation, DatasetFormatVariant, DatasetFormatSpec, DatasetFormatSpecs, DatasetStorageUsage, TrainingMethod, RLHFAlgorithm, AdapterType, TrainingJobStatus, TrainingStatusFilter, TrainingCreateParams, TrainingListParams, TrainingListResponse, TrainingJob, TrainingMetrics, TrainingResourceMetrics, TrainingDeviceMetrics, MetricPoint, MetricGraphConfig, TrainingCheckpoint, TrainingLogs, TrainingLogEntry, TrainingStopResponse, TrainingResumeResponse, CanonicalTrainingRequest, TrainingPreflightDataset, TrainingPreflightWarning, TrainingPreflightResponse, TrainingCapabilityChoice, TrainingConfigFieldCapability, TrainingCapabilities, GPUChoice, WalletBalance, Transaction, TransactionListResponse, TransactionListParams, GPUInfo, GPUPricingResponse, GPUOptionsParams, TrainingRecommendParams, TrainingAdvisorVerdict, TrainingAdvisorRecommendation, TrainingAdvisorUserShape, TrainingAdvisorJustification, TrainingAdvisorBasis, GPUOption, GPUOptionSuggestion, GPUOptionsResponse, InferenceStatus, InferenceCreateParams, InferenceUpdateParams, InferenceDeployment, InferenceDeploymentSummary, InferenceCreateResponse, InferenceListResponse, InferenceUpdateResponse, InferenceLifecycleResponse, InferenceDeleteResponse, InferenceGPUOptionMarket, InferenceGPUOption, InferenceGPUAlternative, InferenceGPUOptionsParams, InferenceGPUOptionsResponse, InferenceCanonicalRequest, InferencePreflightWarning, InferencePreflightResponse, AdapterCompatibility, AdapterCompatibilityResponse, AdapterCompatibilityParams, SupportedArchitecture, ArchitectureScope, SupportedArchitecturesResponse, ApiKey, ApiKeyScope, ApiKeyIntrospection, Organization, OrgMember, OrgInvite, Workspace, Integration, IntegrationCreateParams, StorageUsage, StorageObject, PromptTokensDetails, ChatCompletionUsage, TrainingCadence, TrainingCombinator, ServingKind, TrainingRulePausedReason, LoopTrainingMethod, TrainType, GradersScope, TrainingTrigger, TrainingRunState, TrainingVerdict, TrainingDecision, NotifyState, EvaluationStatus, EvaluationItemStatus, EvaluationWinner, EvaluationWarningCode, ConsentVia, TrainingToolCall, TrainingMessage, TrainingRuleServing, TrainingGPURung, TrainingRule, TrainingRuleConsent, TrainingRulePreflightRefusal, TrainingRuleKeyRef, TrainingRuleKeyCheck, TrainingRulePreflight, TrainingRuleBuildSpec, TrainingRuleTriggerInput, TrainingRuleTrainingInput, TrainingRuleDeployInput, TrainingRuleMoneyInput, TrainingRuleEvaluationInput, TrainingRulePromotionInput, TrainingRuleAcceptTerms, TrainingRulePreflightRequest, TrainingRuleCreateRequest, TrainingRuleUpdateRequest, TrainingRuleConsentRequest, TrainingRuleListParams, TrainingRunPromoteRequest, TrainingRunRejectRequest, TrainingRunRollbackRequest, TrainingRunCancelRequest, TrainingRunListParams, TrainingRun, TrainingRunSummary, TrainingRunEvent, TrainingRunLinks, TrainingRunActions, EvaluationRef, EvaluationDecoding, EvaluationJudgeDimension, EvaluationDimensionScore, EvaluationGraderScore, EvaluationWarning, EvaluationMargin, JudgeAgreementDimension, JudgeAgreement, Evaluation, EvaluationItemSide, EvaluationItem, EvaluationItemListParams, JudgeAgreementParams, AgentSettings, AgentSettingsRequest, InferenceAlias, InferenceAliasRequest, TrainingRuleListResponse, TrainingRuleResponse, TrainingRuleMutationResponse, TrainingRuleDeleteResponse, TrainingRunListResponse, TrainingRunResponse, TrainingRunActionResponse, EvaluationItemsResponse, AgentSettingsResponse, InferenceAliasListResponse, InferenceAliasResponse, InferenceAliasDeleteResponse, } from './types.js';
|
|
79
|
+
/** The closed set of run states a training run never leaves. */
|
|
80
|
+
export { TERMINAL_RUN_STATES } from './types.js';
|
package/dist/index.js
CHANGED
|
@@ -36,7 +36,7 @@ import { Loop } from './resources/loop.js';
|
|
|
36
36
|
* SDK version. Sent as part of the User-Agent header.
|
|
37
37
|
* Must match package.json "version" -- enforced by a contract test.
|
|
38
38
|
*/
|
|
39
|
-
export const VERSION = '0.2.1-dev.
|
|
39
|
+
export const VERSION = '0.2.1-dev.140';
|
|
40
40
|
export class RunBiOS {
|
|
41
41
|
/** Search models, fetch configs, check adapter compatibility. */
|
|
42
42
|
models;
|
|
@@ -103,3 +103,5 @@ export { Training } from './resources/training.js';
|
|
|
103
103
|
export { Wallet } from './resources/wallet.js';
|
|
104
104
|
export { GPU } from './resources/gpu.js';
|
|
105
105
|
export { Inference, validateChatRequest, parseSSE, buildInferenceRequest, contextSizingBasis, CONTEXT_DEFAULT_CEILING, CONTEXT_EDITABLE_FLOOR, } from './resources/inference.js';
|
|
106
|
+
/** The closed set of run states a training run never leaves. */
|
|
107
|
+
export { TERMINAL_RUN_STATES } from './types.js';
|
|
@@ -1,5 +1,5 @@
|
|
|
1
1
|
import type { HttpClient } from '../client.js';
|
|
2
|
-
import type { InferenceDeployment, InferenceDeploymentSummary, InferenceBookingAccepted, InferenceCreateParams, InferenceCreateResponse, InferenceDeleteResponse, InferenceGPUOptionsParams, InferenceGPUOptionsResponse, InferenceLifecycleResponse, InferenceListParams, InferenceListResponse, InferenceNotificationListResponse, InferencePreflightResponse, InferenceUpdateResponse, InferenceUpdateParams } from '../types.js';
|
|
2
|
+
import type { InferenceDeployment, InferenceDeploymentSummary, InferenceBookingAccepted, InferenceCreateParams, InferenceCreateResponse, InferenceDeleteResponse, InferenceGPUOptionsParams, InferenceGPUOptionsResponse, InferenceLifecycleResponse, InferenceListParams, InferenceListResponse, InferenceNotificationListResponse, InferencePreflightResponse, InferenceUpdateResponse, InferenceUpdateParams, InferenceAlias, InferenceAliasRequest, InferenceAliasDeleteResponse } from '../types.js';
|
|
3
3
|
/**
|
|
4
4
|
* Serving context-length policy, owned and enforced by the server. Mirrored
|
|
5
5
|
* here for documentation only -- never to pre-empt a server verdict.
|
|
@@ -213,6 +213,31 @@ export declare class Inference {
|
|
|
213
213
|
restart(id: string): Promise<InferenceLifecycleResponse>;
|
|
214
214
|
update(id: string, params: InferenceUpdateParams): Promise<InferenceUpdateResponse>;
|
|
215
215
|
delete(id: string): Promise<InferenceDeleteResponse>;
|
|
216
|
+
/**
|
|
217
|
+
* Point a name at a deployment, creating the alias or moving an existing one.
|
|
218
|
+
*
|
|
219
|
+
* THIS CHANGES WHAT ANSWERS YOUR TRAFFIC the moment it returns. The reply
|
|
220
|
+
* carries `previous_target_inference_id`, which is what to put back if the
|
|
221
|
+
* new target turns out to be wrong.
|
|
222
|
+
*
|
|
223
|
+
* `origin` records why the handle moved -- a run id, a person, a script --
|
|
224
|
+
* and is worth setting: an alias that changed with no reason recorded is an
|
|
225
|
+
* incident nobody can reconstruct.
|
|
226
|
+
*
|
|
227
|
+
* Refused when the name already belongs to a live deployment
|
|
228
|
+
* (`ALIAS_NAME_IS_A_DEPLOYMENT`), when the target is not running, degraded or
|
|
229
|
+
* provisioning (`TARGET_NOT_SERVABLE`), and with `404` when the target is not
|
|
230
|
+
* in this workspace.
|
|
231
|
+
*/
|
|
232
|
+
setAlias(name: string, params: InferenceAliasRequest): Promise<InferenceAlias>;
|
|
233
|
+
/** Every alias in the workspace, with what each one points at today. */
|
|
234
|
+
listAliases(): Promise<InferenceAlias[]>;
|
|
235
|
+
/**
|
|
236
|
+
* Remove an alias. The name goes back to the deployment that owns it, if one
|
|
237
|
+
* does; callers still using the alias stop resolving, so move it rather than
|
|
238
|
+
* delete it when something is still calling it.
|
|
239
|
+
*/
|
|
240
|
+
deleteAlias(name: string): Promise<InferenceAliasDeleteResponse>;
|
|
216
241
|
/**
|
|
217
242
|
* Model-fit GPU choices joined to the authoritative deployment market
|
|
218
243
|
* snapshot. MODEL-ADDRESSED (recommended, book-first §2): pass `model` (or
|
|
@@ -629,6 +629,51 @@ export class Inference {
|
|
|
629
629
|
delete(id) {
|
|
630
630
|
return this.http.fetchDelete(`/api/inference/${encodeURIComponent(id)}`);
|
|
631
631
|
}
|
|
632
|
+
// --------------------------------------------------------------------------
|
|
633
|
+
// Aliases -- re-pointable public handles
|
|
634
|
+
// --------------------------------------------------------------------------
|
|
635
|
+
//
|
|
636
|
+
// An alias is a name your callers use that you can move to a different
|
|
637
|
+
// deployment without them changing anything. It is what the automatic
|
|
638
|
+
// training loop re-points when it promotes a candidate, and it is what makes
|
|
639
|
+
// a rollback one row write rather than a redeployment.
|
|
640
|
+
//
|
|
641
|
+
// Reads carry `deployments:read` and writes `deployments:write`: an alias
|
|
642
|
+
// decides which model answers a customer's traffic, so it is fenced like the
|
|
643
|
+
// deployment it points at rather than like a label.
|
|
644
|
+
/**
|
|
645
|
+
* Point a name at a deployment, creating the alias or moving an existing one.
|
|
646
|
+
*
|
|
647
|
+
* THIS CHANGES WHAT ANSWERS YOUR TRAFFIC the moment it returns. The reply
|
|
648
|
+
* carries `previous_target_inference_id`, which is what to put back if the
|
|
649
|
+
* new target turns out to be wrong.
|
|
650
|
+
*
|
|
651
|
+
* `origin` records why the handle moved -- a run id, a person, a script --
|
|
652
|
+
* and is worth setting: an alias that changed with no reason recorded is an
|
|
653
|
+
* incident nobody can reconstruct.
|
|
654
|
+
*
|
|
655
|
+
* Refused when the name already belongs to a live deployment
|
|
656
|
+
* (`ALIAS_NAME_IS_A_DEPLOYMENT`), when the target is not running, degraded or
|
|
657
|
+
* provisioning (`TARGET_NOT_SERVABLE`), and with `404` when the target is not
|
|
658
|
+
* in this workspace.
|
|
659
|
+
*/
|
|
660
|
+
async setAlias(name, params) {
|
|
661
|
+
const res = await this.http.fetchPut(`/api/inference/aliases/${encodeURIComponent(name)}`, params);
|
|
662
|
+
return res.alias;
|
|
663
|
+
}
|
|
664
|
+
/** Every alias in the workspace, with what each one points at today. */
|
|
665
|
+
async listAliases() {
|
|
666
|
+
const res = await this.http.fetchGet('/api/inference/aliases');
|
|
667
|
+
return res.aliases || [];
|
|
668
|
+
}
|
|
669
|
+
/**
|
|
670
|
+
* Remove an alias. The name goes back to the deployment that owns it, if one
|
|
671
|
+
* does; callers still using the alias stop resolving, so move it rather than
|
|
672
|
+
* delete it when something is still calling it.
|
|
673
|
+
*/
|
|
674
|
+
async deleteAlias(name) {
|
|
675
|
+
return this.http.fetchDelete(`/api/inference/aliases/${encodeURIComponent(name)}`);
|
|
676
|
+
}
|
|
632
677
|
/**
|
|
633
678
|
* Model-fit GPU choices joined to the authoritative deployment market
|
|
634
679
|
* snapshot. MODEL-ADDRESSED (recommended, book-first §2): pass `model` (or
|
package/dist/resources/loop.d.ts
CHANGED
|
@@ -1,5 +1,5 @@
|
|
|
1
1
|
import type { HttpClient } from '../client.js';
|
|
2
|
-
import type { LoopTrace, LoopTraceListResponse, LoopTraceListParams, LoopCaptureParams, LoopCaptureResult, LoopImportParams, LoopImportResult, LoopSignal, LoopSignalParams, LoopCandidate, LoopCandidateParams, LoopGrader, LoopGraderParams, LoopLabel, LoopLabelCount, LoopGradeResult, LoopDataset, LoopDatasetListParams, LoopDatasetItem, LoopDatasetCreateParams, LoopConfig, LoopConfigParams, LoopBuildRule, LoopBuildRuleParams, LoopStats, LoopJudge, LoopJudgeParams, LoopAgentCredential, LoopAgentStatus, LoopSampleRun, LoopSampleRunParams, LoopJudgeRun, LoopJudgeWork, LoopJudgeVerdict, LoopJudgeVerdictResult } from '../types.js';
|
|
2
|
+
import type { LoopTrace, LoopTraceListResponse, LoopTraceListParams, LoopCaptureParams, LoopCaptureResult, LoopImportParams, LoopImportResult, LoopSignal, LoopSignalParams, LoopCandidate, LoopCandidateParams, LoopGrader, LoopGraderParams, LoopLabel, LoopLabelCount, LoopGradeResult, LoopDataset, LoopDatasetListParams, LoopDatasetItem, LoopDatasetCreateParams, LoopConfig, LoopConfigParams, LoopBuildRule, LoopBuildRuleParams, LoopBuildRuleUpdateParams, LoopStats, LoopJudge, LoopJudgeParams, LoopAgentCredential, LoopAgentStatus, LoopSampleRun, LoopSampleRunParams, LoopJudgeRun, LoopJudgeWork, LoopJudgeVerdict, LoopJudgeVerdictResult, TrainingRule, TrainingRulePreflight, TrainingRulePreflightRequest, TrainingRuleCreateRequest, TrainingRuleUpdateRequest, TrainingRuleConsentRequest, TrainingRuleListParams, TrainingRuleListResponse, TrainingRuleResponse, TrainingRuleMutationResponse, TrainingRuleDeleteResponse, TrainingRun, TrainingRunListParams, TrainingRunListResponse, TrainingRunResponse, TrainingRunActionResponse, TrainingRunPromoteRequest, TrainingRunRejectRequest, TrainingRunRollbackRequest, TrainingRunCancelRequest, Evaluation, EvaluationItemListParams, EvaluationItemsResponse, JudgeAgreement, JudgeAgreementParams, AgentSettings, AgentSettingsRequest } from '../types.js';
|
|
3
3
|
/**
|
|
4
4
|
* The Conscious Loop -- capture what your model was asked and answered, record
|
|
5
5
|
* whether it was right, and turn those judgements into training data.
|
|
@@ -457,4 +457,199 @@ export declare class Loop {
|
|
|
457
457
|
createSampleRun(params: LoopSampleRunParams): Promise<LoopSampleRun>;
|
|
458
458
|
listSampleRuns(): Promise<LoopSampleRun[]>;
|
|
459
459
|
getSampleRun(runId: string): Promise<LoopSampleRun>;
|
|
460
|
+
/**
|
|
461
|
+
* Price a training rule before anyone agrees to it. Writes nothing.
|
|
462
|
+
*
|
|
463
|
+
* Takes the create body without `accept_terms` and answers with the estimate
|
|
464
|
+
* a member has to see first: the pinned model revision, the worst hourly
|
|
465
|
+
* price each GPU ladder can reach, how many hours each ceiling buys, any
|
|
466
|
+
* refusals that would stop a create, the API keys that cannot follow a
|
|
467
|
+
* cutover because they lack `deployments:read`, and `terms_text` -- the
|
|
468
|
+
* exact sentence to show, with real figures in it.
|
|
469
|
+
*
|
|
470
|
+
* Send the `terms_version` it returns back in `createTrainingRule`. Read the
|
|
471
|
+
* figures out of this response rather than inventing ceilings of your own.
|
|
472
|
+
*/
|
|
473
|
+
preflightTrainingRule(params: TrainingRulePreflightRequest): Promise<TrainingRulePreflight>;
|
|
474
|
+
/**
|
|
475
|
+
* Create a training rule and record the consent that pays for it.
|
|
476
|
+
*
|
|
477
|
+
* ACCEPTING THE TERMS AUTHORISES SPENDING FROM YOUR WALLET WHILE YOU ARE NOT
|
|
478
|
+
* PRESENT. From here on the platform may, on its own schedule and without
|
|
479
|
+
* asking again, build a training set, run a training job on rented GPUs,
|
|
480
|
+
* book a second deployment to compare against the one serving your traffic,
|
|
481
|
+
* and pay for the model calls that judge the two. Those charges come out of
|
|
482
|
+
* the wallet of the member whose credentials make this call, up to the
|
|
483
|
+
* ceilings in `money`, and they keep recurring for as long as the rule is
|
|
484
|
+
* enabled. If `promotion.auto_promote` is set, the platform will also
|
|
485
|
+
* re-point your public handle at the new model with nobody reviewing it.
|
|
486
|
+
*
|
|
487
|
+
* `accept_terms` is therefore required, and this method refuses to send the
|
|
488
|
+
* request without it rather than letting the server decide. Call
|
|
489
|
+
* {@link preflightTrainingRule} first, show the person whose wallet pays the
|
|
490
|
+
* `terms_text` and the figures it returns, get an explicit yes, and send the
|
|
491
|
+
* `terms_version` they were shown. Do not invent ceilings or price caps on
|
|
492
|
+
* their behalf.
|
|
493
|
+
*
|
|
494
|
+
* @example
|
|
495
|
+
* ```ts
|
|
496
|
+
* const estimate = await client.loop.preflightTrainingRule(draft);
|
|
497
|
+
* // show estimate.terms_text and the ceilings to the member, get a yes
|
|
498
|
+
* const rule = await client.loop.createTrainingRule({
|
|
499
|
+
* ...draft,
|
|
500
|
+
* accept_terms: { terms_version: estimate.terms_version },
|
|
501
|
+
* });
|
|
502
|
+
* ```
|
|
503
|
+
*/
|
|
504
|
+
createTrainingRule(params: TrainingRuleCreateRequest): Promise<TrainingRule>;
|
|
505
|
+
/**
|
|
506
|
+
* Every training rule in the workspace, with what each one last did and why.
|
|
507
|
+
*
|
|
508
|
+
* `last_reason` and `paused_reason` are the fields worth reading: a rule
|
|
509
|
+
* quiet because it is waiting looks exactly like one quiet because its
|
|
510
|
+
* consent went stale.
|
|
511
|
+
*/
|
|
512
|
+
listTrainingRules(params?: TrainingRuleListParams): Promise<TrainingRuleListResponse>;
|
|
513
|
+
/**
|
|
514
|
+
* Read one rule with its recent runs, what it has spent this month, and how
|
|
515
|
+
* far its judge agrees with your own reviewers.
|
|
516
|
+
*
|
|
517
|
+
* Returned whole rather than unwrapped to the rule: `month_spent_cents` is
|
|
518
|
+
* the number that says whether the monthly ceiling is about to stop it.
|
|
519
|
+
*/
|
|
520
|
+
getTrainingRule(id: string): Promise<TrainingRuleResponse>;
|
|
521
|
+
/**
|
|
522
|
+
* Edit or pause a rule. Absent means unchanged; a null clears the fields
|
|
523
|
+
* that can be cleared.
|
|
524
|
+
*
|
|
525
|
+
* A money-bearing change -- a ceiling, a model, a GPU ladder, the promotion
|
|
526
|
+
* policy -- bumps `revision`, clears the recorded consent and STOPS the rule
|
|
527
|
+
* firing until someone accepts the new amounts. The reply says so in
|
|
528
|
+
* `consent_required`, and carries a fresh `preflight` with the new figures.
|
|
529
|
+
* Pass `expected_revision` to be refused with `REVISION_MISMATCH` rather
|
|
530
|
+
* than overwrite an edit somebody else made in the meantime.
|
|
531
|
+
*/
|
|
532
|
+
updateTrainingRule(id: string, params: TrainingRuleUpdateRequest): Promise<TrainingRuleMutationResponse>;
|
|
533
|
+
/**
|
|
534
|
+
* Accept the rule's current terms, so it may fire again.
|
|
535
|
+
*
|
|
536
|
+
* ACCEPTING THE TERMS AUTHORISES SPENDING FROM YOUR WALLET WHILE YOU ARE NOT
|
|
537
|
+
* PRESENT, on the amounts as they stand right now. This is the same
|
|
538
|
+
* authorisation {@link createTrainingRule} records, given again because a
|
|
539
|
+
* money-bearing edit cleared the old one: training, a candidate deployment
|
|
540
|
+
* and the judge's model calls are charged to the wallet of the member making
|
|
541
|
+
* this call, up to the rule's ceilings, every time it fires.
|
|
542
|
+
*
|
|
543
|
+
* Show the member the current `terms_text` from a fresh
|
|
544
|
+
* {@link preflightTrainingRule} or from the `preflight` on the update reply,
|
|
545
|
+
* and send the `revision` those figures belong to. A stale revision is
|
|
546
|
+
* refused with `409`, which is the point: it means the amounts moved again
|
|
547
|
+
* after they were read.
|
|
548
|
+
*/
|
|
549
|
+
consentTrainingRule(id: string, params: TrainingRuleConsentRequest): Promise<TrainingRule>;
|
|
550
|
+
/**
|
|
551
|
+
* Fire a rule now, without waiting for its cadence.
|
|
552
|
+
*
|
|
553
|
+
* Bypasses the schedule and `min_new_rows` only. The row floors, the money
|
|
554
|
+
* ceilings and the consent all still apply, so this can answer `409
|
|
555
|
+
* CONSENT_REQUIRED` or `422 NOT_ENOUGH_ROWS` with the counts it needed.
|
|
556
|
+
*/
|
|
557
|
+
runTrainingRule(id: string): Promise<TrainingRun>;
|
|
558
|
+
/**
|
|
559
|
+
* Delete a rule. An active run is cancelled; runs that already finished, and
|
|
560
|
+
* anything already promoted, are kept.
|
|
561
|
+
*/
|
|
562
|
+
deleteTrainingRule(id: string): Promise<TrainingRuleDeleteResponse>;
|
|
563
|
+
/**
|
|
564
|
+
* Edit or pause a build rule -- the standing instruction that assembles the
|
|
565
|
+
* training set a training rule then trains on.
|
|
566
|
+
*
|
|
567
|
+
* Changing `spec.deployment_id` while an enabled training rule owns this
|
|
568
|
+
* build rule is refused with `409`: it would silently retrain the next model
|
|
569
|
+
* on a different source's conversations.
|
|
570
|
+
*/
|
|
571
|
+
updateBuildRule(id: string, params: LoopBuildRuleUpdateParams): Promise<LoopBuildRule>;
|
|
572
|
+
/** Firings, newest first. Filter by rule or by the state they are sitting in. */
|
|
573
|
+
listTrainingRuns(params?: TrainingRunListParams): Promise<TrainingRunListResponse>;
|
|
574
|
+
/**
|
|
575
|
+
* One run with its timeline, the addresses of everything it created, and
|
|
576
|
+
* what you may do with it right now.
|
|
577
|
+
*
|
|
578
|
+
* `timeline` is written in the same transaction as each state change, so it
|
|
579
|
+
* is the record of what actually happened rather than a reconstruction.
|
|
580
|
+
* `available_actions` is the honest answer to "can I promote this": a button
|
|
581
|
+
* that cannot work should never be offered.
|
|
582
|
+
*/
|
|
583
|
+
getTrainingRun(id: string): Promise<TrainingRunResponse>;
|
|
584
|
+
/**
|
|
585
|
+
* Promote the candidate by hand: re-point the public handle at the new model.
|
|
586
|
+
*
|
|
587
|
+
* This changes what answers your customers. Pass `expected_revision` to be
|
|
588
|
+
* refused rather than promote against a consent that moved underneath the
|
|
589
|
+
* decision, and `force` only to promote a comparison the evaluation called
|
|
590
|
+
* inconclusive -- without it that case is refused with
|
|
591
|
+
* `INCONCLUSIVE_REQUIRES_FORCE`.
|
|
592
|
+
*/
|
|
593
|
+
promoteTrainingRun(id: string, params?: TrainingRunPromoteRequest): Promise<TrainingRunActionResponse>;
|
|
594
|
+
/** Retire the candidate. What serves your traffic does not change. */
|
|
595
|
+
rejectTrainingRun(id: string, params?: TrainingRunRejectRequest): Promise<TrainingRunActionResponse>;
|
|
596
|
+
/**
|
|
597
|
+
* Undo a promotion, inside the rollback window the run reports in
|
|
598
|
+
* `rollback_available_until` (30 days).
|
|
599
|
+
*
|
|
600
|
+
* The reply carries `serving`: this is the one action besides promotion that
|
|
601
|
+
* changes what answers your traffic, so what it was put back to is returned
|
|
602
|
+
* rather than left to be looked up.
|
|
603
|
+
*/
|
|
604
|
+
rollbackTrainingRun(id: string, params?: TrainingRunRollbackRequest): Promise<TrainingRunActionResponse>;
|
|
605
|
+
/**
|
|
606
|
+
* Stop a run that is still moving. Work already paid for is still billed --
|
|
607
|
+
* cancelling a training job does not refund the hours it burned.
|
|
608
|
+
*/
|
|
609
|
+
cancelTrainingRun(id: string, params?: TrainingRunCancelRequest): Promise<TrainingRunActionResponse>;
|
|
610
|
+
/**
|
|
611
|
+
* The comparison behind a verdict: both models on the same held-out rows,
|
|
612
|
+
* with identical decoding, judge and grader names resolved.
|
|
613
|
+
*
|
|
614
|
+
* Read `warnings` before you read `win_rate`. A win rate over a holdout too
|
|
615
|
+
* small to mean anything, or one measured by a judge that disagrees with
|
|
616
|
+
* your own reviewers, is reported with the warning that says so rather than
|
|
617
|
+
* withheld. `margin_used` is the thresholds this verdict was measured
|
|
618
|
+
* against, frozen with the report, so a later policy change cannot rewrite
|
|
619
|
+
* what a past decision meant.
|
|
620
|
+
*/
|
|
621
|
+
getEvaluation(id: string): Promise<Evaluation>;
|
|
622
|
+
/**
|
|
623
|
+
* The paired conversations behind the numbers: one prompt, both answers, the
|
|
624
|
+
* scores each earned, and which won.
|
|
625
|
+
*
|
|
626
|
+
* Filter by `winner` to read the losses first, which is where a verdict is
|
|
627
|
+
* actually checked. `limit` is capped at 100 by the service.
|
|
628
|
+
*/
|
|
629
|
+
listEvaluationItems(id: string, params?: EvaluationItemListParams): Promise<EvaluationItemsResponse>;
|
|
630
|
+
/**
|
|
631
|
+
* How far a judge agrees with your own reviewers, over the conversations
|
|
632
|
+
* both have scored.
|
|
633
|
+
*
|
|
634
|
+
* This is a gate, not a badge: a rule's `min_judge_agreement` refuses to
|
|
635
|
+
* promote on the word of a judge that does not agree with the people whose
|
|
636
|
+
* product it is. `enough_pairs` is the field to read first -- "not enough
|
|
637
|
+
* reviewer overlap yet" is an answer, and 100% of two pairs is not.
|
|
638
|
+
*/
|
|
639
|
+
getJudgeAgreement(judgeId: string, params?: JudgeAgreementParams): Promise<JudgeAgreement>;
|
|
640
|
+
/**
|
|
641
|
+
* The workspace's agent options: which model it defaults to, the system
|
|
642
|
+
* prompts it judges and samples with, and the monthly cap on what its model
|
|
643
|
+
* calls may spend.
|
|
644
|
+
*/
|
|
645
|
+
getAgentSettings(): Promise<AgentSettings>;
|
|
646
|
+
/**
|
|
647
|
+
* Change them. An absent key leaves that setting exactly where it is; a
|
|
648
|
+
* present null returns it to the platform default. Those are three
|
|
649
|
+
* instructions, not two, so `{}` changes nothing.
|
|
650
|
+
*
|
|
651
|
+
* `eval_monthly_cap_cents` is pushed to the workspace's managed key, so it
|
|
652
|
+
* caps what the agent can spend even if a rule's own ceilings are higher.
|
|
653
|
+
*/
|
|
654
|
+
updateAgentSettings(params: AgentSettingsRequest): Promise<AgentSettings>;
|
|
460
655
|
}
|
package/dist/resources/loop.js
CHANGED
|
@@ -639,4 +639,306 @@ export class Loop {
|
|
|
639
639
|
const res = await this._http.fetchGet(`/api/loop/sample-runs/${encodeURIComponent(runId)}`);
|
|
640
640
|
return res.run;
|
|
641
641
|
}
|
|
642
|
+
// ── automatic training ────────────────────────────────────────────────
|
|
643
|
+
//
|
|
644
|
+
// A training rule is a STANDING INSTRUCTION, not a job. Once it is accepted
|
|
645
|
+
// the platform builds a set, trains a model, books a candidate, compares it
|
|
646
|
+
// against what serves your traffic today and -- if you asked for that --
|
|
647
|
+
// re-points the alias at the winner, on its own schedule, with nobody
|
|
648
|
+
// watching. Every one of those steps spends money from the authorising
|
|
649
|
+
// member's wallet.
|
|
650
|
+
//
|
|
651
|
+
// So the order is always the same: preflight for the estimate and the exact
|
|
652
|
+
// words of the terms, show them to the person whose wallet pays, and only
|
|
653
|
+
// then create with `accept_terms` carrying the `terms_version` they read.
|
|
654
|
+
// Money-bearing edits bump `revision`, clear the consent and stop the rule
|
|
655
|
+
// firing until someone accepts the new amounts.
|
|
656
|
+
//
|
|
657
|
+
// Requires `loop:write` for every mutation, including `preflightTrainingRule`:
|
|
658
|
+
// it is the estimate step of a create, and a key that may not create a rule
|
|
659
|
+
// has no reason to price one.
|
|
660
|
+
/**
|
|
661
|
+
* Price a training rule before anyone agrees to it. Writes nothing.
|
|
662
|
+
*
|
|
663
|
+
* Takes the create body without `accept_terms` and answers with the estimate
|
|
664
|
+
* a member has to see first: the pinned model revision, the worst hourly
|
|
665
|
+
* price each GPU ladder can reach, how many hours each ceiling buys, any
|
|
666
|
+
* refusals that would stop a create, the API keys that cannot follow a
|
|
667
|
+
* cutover because they lack `deployments:read`, and `terms_text` -- the
|
|
668
|
+
* exact sentence to show, with real figures in it.
|
|
669
|
+
*
|
|
670
|
+
* Send the `terms_version` it returns back in `createTrainingRule`. Read the
|
|
671
|
+
* figures out of this response rather than inventing ceilings of your own.
|
|
672
|
+
*/
|
|
673
|
+
async preflightTrainingRule(params) {
|
|
674
|
+
return this._http.fetchPost('/api/loop/training-rules/preflight', params);
|
|
675
|
+
}
|
|
676
|
+
/**
|
|
677
|
+
* Create a training rule and record the consent that pays for it.
|
|
678
|
+
*
|
|
679
|
+
* ACCEPTING THE TERMS AUTHORISES SPENDING FROM YOUR WALLET WHILE YOU ARE NOT
|
|
680
|
+
* PRESENT. From here on the platform may, on its own schedule and without
|
|
681
|
+
* asking again, build a training set, run a training job on rented GPUs,
|
|
682
|
+
* book a second deployment to compare against the one serving your traffic,
|
|
683
|
+
* and pay for the model calls that judge the two. Those charges come out of
|
|
684
|
+
* the wallet of the member whose credentials make this call, up to the
|
|
685
|
+
* ceilings in `money`, and they keep recurring for as long as the rule is
|
|
686
|
+
* enabled. If `promotion.auto_promote` is set, the platform will also
|
|
687
|
+
* re-point your public handle at the new model with nobody reviewing it.
|
|
688
|
+
*
|
|
689
|
+
* `accept_terms` is therefore required, and this method refuses to send the
|
|
690
|
+
* request without it rather than letting the server decide. Call
|
|
691
|
+
* {@link preflightTrainingRule} first, show the person whose wallet pays the
|
|
692
|
+
* `terms_text` and the figures it returns, get an explicit yes, and send the
|
|
693
|
+
* `terms_version` they were shown. Do not invent ceilings or price caps on
|
|
694
|
+
* their behalf.
|
|
695
|
+
*
|
|
696
|
+
* @example
|
|
697
|
+
* ```ts
|
|
698
|
+
* const estimate = await client.loop.preflightTrainingRule(draft);
|
|
699
|
+
* // show estimate.terms_text and the ceilings to the member, get a yes
|
|
700
|
+
* const rule = await client.loop.createTrainingRule({
|
|
701
|
+
* ...draft,
|
|
702
|
+
* accept_terms: { terms_version: estimate.terms_version },
|
|
703
|
+
* });
|
|
704
|
+
* ```
|
|
705
|
+
*/
|
|
706
|
+
async createTrainingRule(params) {
|
|
707
|
+
// Refused here rather than on the wire: a create without a recorded
|
|
708
|
+
// acceptance is a request to spend somebody's money with no record that
|
|
709
|
+
// they agreed, and the SDK should never be the thing that sends it.
|
|
710
|
+
if (!params.accept_terms || !params.accept_terms.terms_version) {
|
|
711
|
+
throw new Error('RunBiOS: createTrainingRule requires accept_terms.terms_version. '
|
|
712
|
+
+ 'Call preflightTrainingRule, show the member terms_text and the ceilings, '
|
|
713
|
+
+ 'and send back the terms_version they accepted -- this rule spends from their wallet.');
|
|
714
|
+
}
|
|
715
|
+
const res = await this._http.fetchPost('/api/loop/training-rules', params);
|
|
716
|
+
return res.rule;
|
|
717
|
+
}
|
|
718
|
+
/**
|
|
719
|
+
* Every training rule in the workspace, with what each one last did and why.
|
|
720
|
+
*
|
|
721
|
+
* `last_reason` and `paused_reason` are the fields worth reading: a rule
|
|
722
|
+
* quiet because it is waiting looks exactly like one quiet because its
|
|
723
|
+
* consent went stale.
|
|
724
|
+
*/
|
|
725
|
+
async listTrainingRules(params = {}) {
|
|
726
|
+
const q = new URLSearchParams();
|
|
727
|
+
if (params.enabled != null)
|
|
728
|
+
q.set('enabled', String(params.enabled));
|
|
729
|
+
if (params.limit != null)
|
|
730
|
+
q.set('limit', String(params.limit));
|
|
731
|
+
if (params.offset != null)
|
|
732
|
+
q.set('offset', String(params.offset));
|
|
733
|
+
const qs = q.toString();
|
|
734
|
+
return this._http.fetchGet(`/api/loop/training-rules${qs ? `?${qs}` : ''}`);
|
|
735
|
+
}
|
|
736
|
+
/**
|
|
737
|
+
* Read one rule with its recent runs, what it has spent this month, and how
|
|
738
|
+
* far its judge agrees with your own reviewers.
|
|
739
|
+
*
|
|
740
|
+
* Returned whole rather than unwrapped to the rule: `month_spent_cents` is
|
|
741
|
+
* the number that says whether the monthly ceiling is about to stop it.
|
|
742
|
+
*/
|
|
743
|
+
async getTrainingRule(id) {
|
|
744
|
+
return this._http.fetchGet(`/api/loop/training-rules/${encodeURIComponent(id)}`);
|
|
745
|
+
}
|
|
746
|
+
/**
|
|
747
|
+
* Edit or pause a rule. Absent means unchanged; a null clears the fields
|
|
748
|
+
* that can be cleared.
|
|
749
|
+
*
|
|
750
|
+
* A money-bearing change -- a ceiling, a model, a GPU ladder, the promotion
|
|
751
|
+
* policy -- bumps `revision`, clears the recorded consent and STOPS the rule
|
|
752
|
+
* firing until someone accepts the new amounts. The reply says so in
|
|
753
|
+
* `consent_required`, and carries a fresh `preflight` with the new figures.
|
|
754
|
+
* Pass `expected_revision` to be refused with `REVISION_MISMATCH` rather
|
|
755
|
+
* than overwrite an edit somebody else made in the meantime.
|
|
756
|
+
*/
|
|
757
|
+
async updateTrainingRule(id, params) {
|
|
758
|
+
return this._http.fetchPut(`/api/loop/training-rules/${encodeURIComponent(id)}`, params);
|
|
759
|
+
}
|
|
760
|
+
/**
|
|
761
|
+
* Accept the rule's current terms, so it may fire again.
|
|
762
|
+
*
|
|
763
|
+
* ACCEPTING THE TERMS AUTHORISES SPENDING FROM YOUR WALLET WHILE YOU ARE NOT
|
|
764
|
+
* PRESENT, on the amounts as they stand right now. This is the same
|
|
765
|
+
* authorisation {@link createTrainingRule} records, given again because a
|
|
766
|
+
* money-bearing edit cleared the old one: training, a candidate deployment
|
|
767
|
+
* and the judge's model calls are charged to the wallet of the member making
|
|
768
|
+
* this call, up to the rule's ceilings, every time it fires.
|
|
769
|
+
*
|
|
770
|
+
* Show the member the current `terms_text` from a fresh
|
|
771
|
+
* {@link preflightTrainingRule} or from the `preflight` on the update reply,
|
|
772
|
+
* and send the `revision` those figures belong to. A stale revision is
|
|
773
|
+
* refused with `409`, which is the point: it means the amounts moved again
|
|
774
|
+
* after they were read.
|
|
775
|
+
*/
|
|
776
|
+
async consentTrainingRule(id, params) {
|
|
777
|
+
const res = await this._http.fetchPost(`/api/loop/training-rules/${encodeURIComponent(id)}/consent`, params);
|
|
778
|
+
return res.rule;
|
|
779
|
+
}
|
|
780
|
+
/**
|
|
781
|
+
* Fire a rule now, without waiting for its cadence.
|
|
782
|
+
*
|
|
783
|
+
* Bypasses the schedule and `min_new_rows` only. The row floors, the money
|
|
784
|
+
* ceilings and the consent all still apply, so this can answer `409
|
|
785
|
+
* CONSENT_REQUIRED` or `422 NOT_ENOUGH_ROWS` with the counts it needed.
|
|
786
|
+
*/
|
|
787
|
+
async runTrainingRule(id) {
|
|
788
|
+
const res = await this._http.fetchPost(`/api/loop/training-rules/${encodeURIComponent(id)}/run`, {});
|
|
789
|
+
return res.run;
|
|
790
|
+
}
|
|
791
|
+
/**
|
|
792
|
+
* Delete a rule. An active run is cancelled; runs that already finished, and
|
|
793
|
+
* anything already promoted, are kept.
|
|
794
|
+
*/
|
|
795
|
+
async deleteTrainingRule(id) {
|
|
796
|
+
return this._http.fetchDelete(`/api/loop/training-rules/${encodeURIComponent(id)}`);
|
|
797
|
+
}
|
|
798
|
+
/**
|
|
799
|
+
* Edit or pause a build rule -- the standing instruction that assembles the
|
|
800
|
+
* training set a training rule then trains on.
|
|
801
|
+
*
|
|
802
|
+
* Changing `spec.deployment_id` while an enabled training rule owns this
|
|
803
|
+
* build rule is refused with `409`: it would silently retrain the next model
|
|
804
|
+
* on a different source's conversations.
|
|
805
|
+
*/
|
|
806
|
+
async updateBuildRule(id, params) {
|
|
807
|
+
const res = await this._http.fetchPut(`/api/loop/build-rules/${encodeURIComponent(id)}`, params);
|
|
808
|
+
return res.rule;
|
|
809
|
+
}
|
|
810
|
+
// ── training runs ─────────────────────────────────────────────────────
|
|
811
|
+
/** Firings, newest first. Filter by rule or by the state they are sitting in. */
|
|
812
|
+
async listTrainingRuns(params = {}) {
|
|
813
|
+
const q = new URLSearchParams();
|
|
814
|
+
if (params.rule_id)
|
|
815
|
+
q.set('rule_id', params.rule_id);
|
|
816
|
+
if (params.state)
|
|
817
|
+
q.set('state', params.state);
|
|
818
|
+
if (params.limit != null)
|
|
819
|
+
q.set('limit', String(params.limit));
|
|
820
|
+
if (params.offset != null)
|
|
821
|
+
q.set('offset', String(params.offset));
|
|
822
|
+
const qs = q.toString();
|
|
823
|
+
return this._http.fetchGet(`/api/loop/training-runs${qs ? `?${qs}` : ''}`);
|
|
824
|
+
}
|
|
825
|
+
/**
|
|
826
|
+
* One run with its timeline, the addresses of everything it created, and
|
|
827
|
+
* what you may do with it right now.
|
|
828
|
+
*
|
|
829
|
+
* `timeline` is written in the same transaction as each state change, so it
|
|
830
|
+
* is the record of what actually happened rather than a reconstruction.
|
|
831
|
+
* `available_actions` is the honest answer to "can I promote this": a button
|
|
832
|
+
* that cannot work should never be offered.
|
|
833
|
+
*/
|
|
834
|
+
async getTrainingRun(id) {
|
|
835
|
+
return this._http.fetchGet(`/api/loop/training-runs/${encodeURIComponent(id)}`);
|
|
836
|
+
}
|
|
837
|
+
/**
|
|
838
|
+
* Promote the candidate by hand: re-point the public handle at the new model.
|
|
839
|
+
*
|
|
840
|
+
* This changes what answers your customers. Pass `expected_revision` to be
|
|
841
|
+
* refused rather than promote against a consent that moved underneath the
|
|
842
|
+
* decision, and `force` only to promote a comparison the evaluation called
|
|
843
|
+
* inconclusive -- without it that case is refused with
|
|
844
|
+
* `INCONCLUSIVE_REQUIRES_FORCE`.
|
|
845
|
+
*/
|
|
846
|
+
async promoteTrainingRun(id, params = {}) {
|
|
847
|
+
return this._http.fetchPost(`/api/loop/training-runs/${encodeURIComponent(id)}/promote`, params);
|
|
848
|
+
}
|
|
849
|
+
/** Retire the candidate. What serves your traffic does not change. */
|
|
850
|
+
async rejectTrainingRun(id, params = {}) {
|
|
851
|
+
return this._http.fetchPost(`/api/loop/training-runs/${encodeURIComponent(id)}/reject`, params);
|
|
852
|
+
}
|
|
853
|
+
/**
|
|
854
|
+
* Undo a promotion, inside the rollback window the run reports in
|
|
855
|
+
* `rollback_available_until` (30 days).
|
|
856
|
+
*
|
|
857
|
+
* The reply carries `serving`: this is the one action besides promotion that
|
|
858
|
+
* changes what answers your traffic, so what it was put back to is returned
|
|
859
|
+
* rather than left to be looked up.
|
|
860
|
+
*/
|
|
861
|
+
async rollbackTrainingRun(id, params = {}) {
|
|
862
|
+
return this._http.fetchPost(`/api/loop/training-runs/${encodeURIComponent(id)}/rollback`, params);
|
|
863
|
+
}
|
|
864
|
+
/**
|
|
865
|
+
* Stop a run that is still moving. Work already paid for is still billed --
|
|
866
|
+
* cancelling a training job does not refund the hours it burned.
|
|
867
|
+
*/
|
|
868
|
+
async cancelTrainingRun(id, params = {}) {
|
|
869
|
+
return this._http.fetchPost(`/api/loop/training-runs/${encodeURIComponent(id)}/cancel`, params);
|
|
870
|
+
}
|
|
871
|
+
// ── the comparison report ─────────────────────────────────────────────
|
|
872
|
+
/**
|
|
873
|
+
* The comparison behind a verdict: both models on the same held-out rows,
|
|
874
|
+
* with identical decoding, judge and grader names resolved.
|
|
875
|
+
*
|
|
876
|
+
* Read `warnings` before you read `win_rate`. A win rate over a holdout too
|
|
877
|
+
* small to mean anything, or one measured by a judge that disagrees with
|
|
878
|
+
* your own reviewers, is reported with the warning that says so rather than
|
|
879
|
+
* withheld. `margin_used` is the thresholds this verdict was measured
|
|
880
|
+
* against, frozen with the report, so a later policy change cannot rewrite
|
|
881
|
+
* what a past decision meant.
|
|
882
|
+
*/
|
|
883
|
+
async getEvaluation(id) {
|
|
884
|
+
return this._http.fetchGet(`/api/loop/evaluations/${encodeURIComponent(id)}`);
|
|
885
|
+
}
|
|
886
|
+
/**
|
|
887
|
+
* The paired conversations behind the numbers: one prompt, both answers, the
|
|
888
|
+
* scores each earned, and which won.
|
|
889
|
+
*
|
|
890
|
+
* Filter by `winner` to read the losses first, which is where a verdict is
|
|
891
|
+
* actually checked. `limit` is capped at 100 by the service.
|
|
892
|
+
*/
|
|
893
|
+
async listEvaluationItems(id, params = {}) {
|
|
894
|
+
const q = new URLSearchParams();
|
|
895
|
+
if (params.winner)
|
|
896
|
+
q.set('winner', params.winner);
|
|
897
|
+
if (params.limit != null)
|
|
898
|
+
q.set('limit', String(params.limit));
|
|
899
|
+
if (params.offset != null)
|
|
900
|
+
q.set('offset', String(params.offset));
|
|
901
|
+
const qs = q.toString();
|
|
902
|
+
return this._http.fetchGet(`/api/loop/evaluations/${encodeURIComponent(id)}/items${qs ? `?${qs}` : ''}`);
|
|
903
|
+
}
|
|
904
|
+
/**
|
|
905
|
+
* How far a judge agrees with your own reviewers, over the conversations
|
|
906
|
+
* both have scored.
|
|
907
|
+
*
|
|
908
|
+
* This is a gate, not a badge: a rule's `min_judge_agreement` refuses to
|
|
909
|
+
* promote on the word of a judge that does not agree with the people whose
|
|
910
|
+
* product it is. `enough_pairs` is the field to read first -- "not enough
|
|
911
|
+
* reviewer overlap yet" is an answer, and 100% of two pairs is not.
|
|
912
|
+
*/
|
|
913
|
+
async getJudgeAgreement(judgeId, params = {}) {
|
|
914
|
+
const q = new URLSearchParams();
|
|
915
|
+
if (params.from)
|
|
916
|
+
q.set('from', params.from);
|
|
917
|
+
if (params.to)
|
|
918
|
+
q.set('to', params.to);
|
|
919
|
+
const qs = q.toString();
|
|
920
|
+
return this._http.fetchGet(`/api/loop/judges/${encodeURIComponent(judgeId)}/agreement${qs ? `?${qs}` : ''}`);
|
|
921
|
+
}
|
|
922
|
+
// ── agent settings ────────────────────────────────────────────────────
|
|
923
|
+
/**
|
|
924
|
+
* The workspace's agent options: which model it defaults to, the system
|
|
925
|
+
* prompts it judges and samples with, and the monthly cap on what its model
|
|
926
|
+
* calls may spend.
|
|
927
|
+
*/
|
|
928
|
+
async getAgentSettings() {
|
|
929
|
+
const res = await this._http.fetchGet('/api/loop/agent/settings');
|
|
930
|
+
return res.settings;
|
|
931
|
+
}
|
|
932
|
+
/**
|
|
933
|
+
* Change them. An absent key leaves that setting exactly where it is; a
|
|
934
|
+
* present null returns it to the platform default. Those are three
|
|
935
|
+
* instructions, not two, so `{}` changes nothing.
|
|
936
|
+
*
|
|
937
|
+
* `eval_monthly_cap_cents` is pushed to the workspace's managed key, so it
|
|
938
|
+
* caps what the agent can spend even if a rule's own ceilings are higher.
|
|
939
|
+
*/
|
|
940
|
+
async updateAgentSettings(params) {
|
|
941
|
+
const res = await this._http.fetchPut('/api/loop/agent/settings', params);
|
|
942
|
+
return res.settings;
|
|
943
|
+
}
|
|
642
944
|
}
|