runbios-sdk 0.2.1-dev.137 → 0.2.1-dev.141
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/dist/index.d.ts +4 -2
- package/dist/index.js +3 -1
- package/dist/resources/inference.d.ts +26 -1
- package/dist/resources/inference.js +45 -0
- package/dist/resources/loop.d.ts +196 -1
- package/dist/resources/loop.js +302 -0
- package/dist/types.d.ts +685 -0
- package/dist/types.js +11 -1
- package/package.json +1 -1
package/dist/types.d.ts
CHANGED
|
@@ -2783,3 +2783,688 @@ export interface LoopLabelCount {
|
|
|
2783
2783
|
traces: number;
|
|
2784
2784
|
key: string;
|
|
2785
2785
|
}
|
|
2786
|
+
export type TrainingCadence = 'none' | 'daily' | 'weekly';
|
|
2787
|
+
export type TrainingCombinator = 'and' | 'or';
|
|
2788
|
+
/** What answers the customer's traffic today, and therefore what promotion re-points. */
|
|
2789
|
+
export type ServingKind = 'serverless_slug' | 'deployment';
|
|
2790
|
+
export type TrainingRulePausedReason = 'consent_invalid' | 'insufficient_funds' | 'authorizer_not_member' | 'model_not_trainable' | 'platform_capacity'
|
|
2791
|
+
/** A peer was unreachable when the rule was saved, so it has not been validated yet. */
|
|
2792
|
+
| 'validation_pending';
|
|
2793
|
+
/**
|
|
2794
|
+
* How a training rule trains, on the wire of the automatic-training routes.
|
|
2795
|
+
*
|
|
2796
|
+
* DELIBERATELY NOT {@link TrainingMethod}. That union -- 'sft' | 'pt' -- is
|
|
2797
|
+
* training-service's, and the two services enumerate different things: a
|
|
2798
|
+
* training job can be a plain pre-training run, and a training rule cannot,
|
|
2799
|
+
* while a rule may ask for preference training and a job asks for that through
|
|
2800
|
+
* a different field. The values here are the loop service's own constants
|
|
2801
|
+
* (TrainingMethodSFT / TrainingMethodRLHF in
|
|
2802
|
+
* services/loop-service/cmd/training_types.go), which is what these routes
|
|
2803
|
+
* accept and return. Sharing the training-service union here advertised 'pt',
|
|
2804
|
+
* which this service refuses, and made 'rlhf', which it returns, unspellable.
|
|
2805
|
+
*
|
|
2806
|
+
* `rlhf_type` stays a plain string on purpose: the platform refuses it until
|
|
2807
|
+
* capabilities enable preference training, and an SDK union would have to be
|
|
2808
|
+
* republished to keep up with a server-side capability flag.
|
|
2809
|
+
*/
|
|
2810
|
+
export type LoopTrainingMethod = 'sft' | 'rlhf';
|
|
2811
|
+
export type TrainType = 'lora' | 'qlora' | 'full';
|
|
2812
|
+
export type GradersScope = 'source' | 'all' | 'none';
|
|
2813
|
+
export type TrainingTrigger = 'rows' | 'cadence' | 'both' | 'manual';
|
|
2814
|
+
export type TrainingRunState = 'built' | 'registering' | 'dataset_validating' | 'preflighting' | 'job_creating' | 'training' | 'checkpoint_ready' | 'candidate_booking' | 'candidate_running' | 'evaluating' | 'reported' | 'awaiting_review' | 'promoting' | 'waiting_funds' | 'promoted' | 'rejected' | 'expired' | 'failed' | 'budget_stopped' | 'cancelled' | 'superseded' | 'rolled_back';
|
|
2815
|
+
/** The closed set a run never leaves. */
|
|
2816
|
+
export declare const TERMINAL_RUN_STATES: readonly TrainingRunState[];
|
|
2817
|
+
export type TrainingVerdict = 'better' | 'not_better' | 'inconclusive' | 'not_evaluated';
|
|
2818
|
+
export type TrainingDecision = 'auto_promoted' | 'promoted' | 'rejected' | 'auto_rejected' | 'rolled_back';
|
|
2819
|
+
export type NotifyState = 'none' | 'pending' | 'sending' | 'sent' | 'suppressed' | 'failed';
|
|
2820
|
+
export type EvaluationStatus = 'open' | 'done' | 'failed' | 'budget_stopped';
|
|
2821
|
+
export type EvaluationItemStatus = 'pending' | 'working' | 'done' | 'failed';
|
|
2822
|
+
export type EvaluationWinner = 'candidate' | 'incumbent' | 'tie';
|
|
2823
|
+
/** Typed warning codes. Each surface renders its own plain sentence. */
|
|
2824
|
+
export type EvaluationWarningCode = 'holdout_too_small' | 'judge_unreliable' | 'graders_not_applied' | 'many_failures' | 'no_judge' | 'judge_labelled_training_rows' | 'capture_source_moved';
|
|
2825
|
+
export type ConsentVia = 'console' | 'api_key' | 'sdk' | 'mcp';
|
|
2826
|
+
/** A function an assistant turn asked to invoke. `arguments` is JSON-encoded. */
|
|
2827
|
+
export interface TrainingToolCall {
|
|
2828
|
+
id?: string;
|
|
2829
|
+
type?: string;
|
|
2830
|
+
function: {
|
|
2831
|
+
name: string;
|
|
2832
|
+
arguments: string;
|
|
2833
|
+
};
|
|
2834
|
+
}
|
|
2835
|
+
/** One conversational turn, in the chat-completions shape. */
|
|
2836
|
+
export interface TrainingMessage {
|
|
2837
|
+
role: string;
|
|
2838
|
+
content?: string;
|
|
2839
|
+
tool_calls?: TrainingToolCall[];
|
|
2840
|
+
tool_call_id?: string;
|
|
2841
|
+
name?: string;
|
|
2842
|
+
}
|
|
2843
|
+
/** What serves this model today. The name is also the alias a promotion writes. */
|
|
2844
|
+
export interface TrainingRuleServing {
|
|
2845
|
+
kind: ServingKind;
|
|
2846
|
+
name: string;
|
|
2847
|
+
/** Set only when kind is 'deployment'. */
|
|
2848
|
+
deployment_id: string | null;
|
|
2849
|
+
}
|
|
2850
|
+
/** One rung of a ranked GPU ladder. The provider is the neutral public brand. */
|
|
2851
|
+
export interface TrainingGPURung {
|
|
2852
|
+
gpu_type: string;
|
|
2853
|
+
gpu_count: number;
|
|
2854
|
+
provider: string;
|
|
2855
|
+
region: string;
|
|
2856
|
+
tier: string;
|
|
2857
|
+
}
|
|
2858
|
+
/** The consent object: one standing instruction to train, judge and possibly promote. */
|
|
2859
|
+
export interface TrainingRule {
|
|
2860
|
+
id: string;
|
|
2861
|
+
workspace_id: string;
|
|
2862
|
+
build_rule_id: string;
|
|
2863
|
+
name: string;
|
|
2864
|
+
enabled: boolean;
|
|
2865
|
+
/** Why the platform stopped firing this rule. Waiting must not look like broken. */
|
|
2866
|
+
paused_reason: TrainingRulePausedReason | null;
|
|
2867
|
+
cadence: TrainingCadence;
|
|
2868
|
+
cadence_hour_utc: number;
|
|
2869
|
+
cadence_weekday: number | null;
|
|
2870
|
+
combinator: TrainingCombinator;
|
|
2871
|
+
min_new_rows: number | null;
|
|
2872
|
+
/** Spreads firings across the hour so every daily rule does not land on one minute. */
|
|
2873
|
+
jitter_seconds: number;
|
|
2874
|
+
next_due_at: string | null;
|
|
2875
|
+
last_checked_at: string | null;
|
|
2876
|
+
last_fired_at: string | null;
|
|
2877
|
+
last_run_id: string | null;
|
|
2878
|
+
/** Rendered verbatim: "waiting: ...", "due, but ...", "fired: run 7 from ...". */
|
|
2879
|
+
last_reason: string | null;
|
|
2880
|
+
serving: TrainingRuleServing;
|
|
2881
|
+
model_id: string;
|
|
2882
|
+
/** A 40-hex commit, pinned when the rule is saved. */
|
|
2883
|
+
model_revision: string;
|
|
2884
|
+
training_method: LoopTrainingMethod;
|
|
2885
|
+
/** Refused by the platform until capabilities enable preference training. */
|
|
2886
|
+
rlhf_type: string | null;
|
|
2887
|
+
train_type: TrainType;
|
|
2888
|
+
config: Record<string, unknown>;
|
|
2889
|
+
train_gpu_priorities: TrainingGPURung[];
|
|
2890
|
+
train_max_price_hour_cents: number;
|
|
2891
|
+
deploy_gpu_priorities: TrainingGPURung[];
|
|
2892
|
+
deploy_max_price_hour_cents: number;
|
|
2893
|
+
training_ceiling_cents: number;
|
|
2894
|
+
candidate_ceiling_cents: number;
|
|
2895
|
+
/** A dollar figure, never a call count. */
|
|
2896
|
+
eval_ceiling_cents: number;
|
|
2897
|
+
monthly_ceiling_cents: number | null;
|
|
2898
|
+
eval_judge_id: string | null;
|
|
2899
|
+
eval_judge_model: string | null;
|
|
2900
|
+
eval_graders_scope: GradersScope;
|
|
2901
|
+
eval_max_rows: number;
|
|
2902
|
+
eval_max_tokens: number;
|
|
2903
|
+
min_holdout_rows: number;
|
|
2904
|
+
auto_promote: boolean;
|
|
2905
|
+
promote_margin: number;
|
|
2906
|
+
promote_min_win_rate: number;
|
|
2907
|
+
/** A gate, not a badge. */
|
|
2908
|
+
min_judge_agreement: number;
|
|
2909
|
+
review_window_hours: number;
|
|
2910
|
+
candidate_boot_deadline_minutes: number;
|
|
2911
|
+
keep_candidate_warm_minutes: number;
|
|
2912
|
+
/** The member whose wallet pays, and the identity every peer call is stamped with. */
|
|
2913
|
+
authorized_by: string;
|
|
2914
|
+
terms_version: string;
|
|
2915
|
+
revision: number;
|
|
2916
|
+
/** Equal to revision only while the consent is current. */
|
|
2917
|
+
accepted_revision: number;
|
|
2918
|
+
accepted_at: string | null;
|
|
2919
|
+
deleted_at: string | null;
|
|
2920
|
+
created_at: string;
|
|
2921
|
+
updated_at: string;
|
|
2922
|
+
}
|
|
2923
|
+
/** One append-only record of a member agreeing to spend, with the words they read. */
|
|
2924
|
+
export interface TrainingRuleConsent {
|
|
2925
|
+
id: number;
|
|
2926
|
+
rule_id: string;
|
|
2927
|
+
workspace_id: string;
|
|
2928
|
+
revision: number;
|
|
2929
|
+
terms_version: string;
|
|
2930
|
+
terms_text: string;
|
|
2931
|
+
snapshot: Record<string, unknown>;
|
|
2932
|
+
accepted_by: string;
|
|
2933
|
+
accepted_via: ConsentVia;
|
|
2934
|
+
accepted_at: string;
|
|
2935
|
+
}
|
|
2936
|
+
export interface TrainingRulePreflightRefusal {
|
|
2937
|
+
stage: string;
|
|
2938
|
+
code: string;
|
|
2939
|
+
message: string;
|
|
2940
|
+
}
|
|
2941
|
+
/** One key that cannot follow a cutover, named so the caller can say which. */
|
|
2942
|
+
export interface TrainingRuleKeyRef {
|
|
2943
|
+
key_id: string;
|
|
2944
|
+
prefix: string;
|
|
2945
|
+
name: string;
|
|
2946
|
+
}
|
|
2947
|
+
export interface TrainingRuleKeyCheck {
|
|
2948
|
+
keys_missing_deployments_read: TrainingRuleKeyRef[];
|
|
2949
|
+
}
|
|
2950
|
+
/** The side-effect-free estimate, and the exact sentence the member will accept. */
|
|
2951
|
+
export interface TrainingRulePreflight {
|
|
2952
|
+
valid: boolean;
|
|
2953
|
+
model_revision: string;
|
|
2954
|
+
worst_hourly_training_cents: number;
|
|
2955
|
+
worst_hourly_candidate_cents: number;
|
|
2956
|
+
max_training_hours: number;
|
|
2957
|
+
max_candidate_hours: number;
|
|
2958
|
+
/** Informational. The fence on evaluation is the dollar ceiling, never this. */
|
|
2959
|
+
eval_calls_max: number;
|
|
2960
|
+
training_ceiling_cents: number;
|
|
2961
|
+
candidate_ceiling_cents: number;
|
|
2962
|
+
eval_ceiling_cents: number;
|
|
2963
|
+
warnings: string[];
|
|
2964
|
+
refusals: TrainingRulePreflightRefusal[];
|
|
2965
|
+
key_check: TrainingRuleKeyCheck;
|
|
2966
|
+
terms_text: string;
|
|
2967
|
+
terms_version: string;
|
|
2968
|
+
}
|
|
2969
|
+
export interface TrainingRuleBuildSpec {
|
|
2970
|
+
method: string;
|
|
2971
|
+
spec: Record<string, unknown>;
|
|
2972
|
+
}
|
|
2973
|
+
/**
|
|
2974
|
+
* The editable half of a build rule.
|
|
2975
|
+
*
|
|
2976
|
+
* Deliberately not `LoopBuildRuleParams`: that type requires `name` and
|
|
2977
|
+
* `method`, which the PUT route does not accept, so an edit written against it
|
|
2978
|
+
* would have to resend two fields the platform ignores and a caller could not
|
|
2979
|
+
* tell that changing them did nothing.
|
|
2980
|
+
*/
|
|
2981
|
+
export interface LoopBuildRuleUpdateParams {
|
|
2982
|
+
enabled?: boolean;
|
|
2983
|
+
/** Rows reviewed SINCE THE LAST BUILD before this fires again. Minimum 10. */
|
|
2984
|
+
min_new_rows?: number;
|
|
2985
|
+
/** The same selection `createDataset` takes, replayed verbatim. */
|
|
2986
|
+
spec?: LoopDatasetCreateParams;
|
|
2987
|
+
}
|
|
2988
|
+
/**
|
|
2989
|
+
* `?` means absent: leave that part of the sentence alone. `null` is only
|
|
2990
|
+
* allowed where the column is nullable and clearing it is a real edit, and it
|
|
2991
|
+
* is spelled out field by field rather than applied to the whole object.
|
|
2992
|
+
*/
|
|
2993
|
+
export interface TrainingRuleTriggerInput {
|
|
2994
|
+
cadence?: TrainingCadence;
|
|
2995
|
+
cadence_hour_utc?: number;
|
|
2996
|
+
/** null drops the weekday, which is what a weekly rule moved to daily needs. */
|
|
2997
|
+
cadence_weekday?: number | null;
|
|
2998
|
+
combinator?: TrainingCombinator;
|
|
2999
|
+
/** null removes the row floor, leaving the schedule as the only trigger. */
|
|
3000
|
+
min_new_rows?: number | null;
|
|
3001
|
+
}
|
|
3002
|
+
export interface TrainingRuleTrainingInput {
|
|
3003
|
+
model_id?: string;
|
|
3004
|
+
model_revision?: string;
|
|
3005
|
+
training_method?: LoopTrainingMethod;
|
|
3006
|
+
/**
|
|
3007
|
+
* A plain string, refused by the platform until capabilities enable
|
|
3008
|
+
* preference training. null clears it, which is what moving a rule back to
|
|
3009
|
+
* plain supervised training means.
|
|
3010
|
+
*/
|
|
3011
|
+
rlhf_type?: string | null;
|
|
3012
|
+
train_type?: TrainType;
|
|
3013
|
+
config?: Record<string, unknown>;
|
|
3014
|
+
train_gpu_priorities?: TrainingGPURung[];
|
|
3015
|
+
train_max_price_hour_cents?: number;
|
|
3016
|
+
}
|
|
3017
|
+
export interface TrainingRuleDeployInput {
|
|
3018
|
+
deploy_gpu_priorities?: TrainingGPURung[];
|
|
3019
|
+
deploy_max_price_hour_cents?: number;
|
|
3020
|
+
context_length?: number;
|
|
3021
|
+
quant?: string;
|
|
3022
|
+
serving_config?: Record<string, unknown>;
|
|
3023
|
+
hf_integration_id?: string;
|
|
3024
|
+
}
|
|
3025
|
+
export interface TrainingRuleMoneyInput {
|
|
3026
|
+
training_ceiling_cents?: number;
|
|
3027
|
+
candidate_ceiling_cents?: number;
|
|
3028
|
+
eval_ceiling_cents?: number;
|
|
3029
|
+
/** null removes the monthly cap. Absent leaves it exactly where it stands. */
|
|
3030
|
+
monthly_ceiling_cents?: number | null;
|
|
3031
|
+
eval_max_rows?: number;
|
|
3032
|
+
}
|
|
3033
|
+
export interface TrainingRuleEvaluationInput {
|
|
3034
|
+
/** null removes the judge from this rule. */
|
|
3035
|
+
judge_id?: string | null;
|
|
3036
|
+
/** null drops the override, returning to the judge's own advisory model. */
|
|
3037
|
+
judge_model?: string | null;
|
|
3038
|
+
graders_scope?: GradersScope;
|
|
3039
|
+
min_holdout_rows?: number;
|
|
3040
|
+
eval_max_tokens?: number;
|
|
3041
|
+
}
|
|
3042
|
+
export interface TrainingRulePromotionInput {
|
|
3043
|
+
auto_promote?: boolean;
|
|
3044
|
+
promote_margin?: number;
|
|
3045
|
+
promote_min_win_rate?: number;
|
|
3046
|
+
min_judge_agreement?: number;
|
|
3047
|
+
review_window_hours?: number;
|
|
3048
|
+
keep_candidate_warm_minutes?: number;
|
|
3049
|
+
}
|
|
3050
|
+
/**
|
|
3051
|
+
* The version of the terms the member read and accepted.
|
|
3052
|
+
*
|
|
3053
|
+
* Send back the `terms_version` the preflight returned. The SDK deliberately
|
|
3054
|
+
* ships no constant for it: a pinned version in a published package goes stale
|
|
3055
|
+
* the moment the platform revises the terms, and the value a member consented
|
|
3056
|
+
* to has to be the one they were actually shown.
|
|
3057
|
+
*/
|
|
3058
|
+
export interface TrainingRuleAcceptTerms {
|
|
3059
|
+
terms_version: string;
|
|
3060
|
+
}
|
|
3061
|
+
/** The create body without the yes. The preflight route takes exactly this. */
|
|
3062
|
+
export interface TrainingRulePreflightRequest {
|
|
3063
|
+
workspace_id?: string;
|
|
3064
|
+
name?: string;
|
|
3065
|
+
/** Exactly one of build_rule_id and build is given. */
|
|
3066
|
+
build_rule_id?: string;
|
|
3067
|
+
build?: TrainingRuleBuildSpec;
|
|
3068
|
+
trigger?: TrainingRuleTriggerInput;
|
|
3069
|
+
serving?: TrainingRuleServing;
|
|
3070
|
+
training?: TrainingRuleTrainingInput;
|
|
3071
|
+
deploy?: TrainingRuleDeployInput;
|
|
3072
|
+
money?: TrainingRuleMoneyInput;
|
|
3073
|
+
evaluation?: TrainingRuleEvaluationInput;
|
|
3074
|
+
promotion?: TrainingRulePromotionInput;
|
|
3075
|
+
enabled?: boolean;
|
|
3076
|
+
}
|
|
3077
|
+
export interface TrainingRuleCreateRequest extends TrainingRulePreflightRequest {
|
|
3078
|
+
/** Absent is a refusal, not a default. */
|
|
3079
|
+
accept_terms?: TrainingRuleAcceptTerms;
|
|
3080
|
+
}
|
|
3081
|
+
export interface TrainingRuleUpdateRequest extends TrainingRulePreflightRequest {
|
|
3082
|
+
expected_revision?: number;
|
|
3083
|
+
accept_terms?: TrainingRuleAcceptTerms;
|
|
3084
|
+
}
|
|
3085
|
+
export interface TrainingRuleConsentRequest {
|
|
3086
|
+
terms_version: string;
|
|
3087
|
+
revision: number;
|
|
3088
|
+
}
|
|
3089
|
+
export interface TrainingRuleListParams {
|
|
3090
|
+
enabled?: boolean;
|
|
3091
|
+
limit?: number;
|
|
3092
|
+
offset?: number;
|
|
3093
|
+
}
|
|
3094
|
+
export interface TrainingRunPromoteRequest {
|
|
3095
|
+
expected_revision?: number;
|
|
3096
|
+
/** Required to promote an inconclusive comparison. */
|
|
3097
|
+
force?: boolean;
|
|
3098
|
+
}
|
|
3099
|
+
export interface TrainingRunRejectRequest {
|
|
3100
|
+
reason?: string;
|
|
3101
|
+
}
|
|
3102
|
+
export interface TrainingRunRollbackRequest {
|
|
3103
|
+
reason?: string;
|
|
3104
|
+
}
|
|
3105
|
+
export interface TrainingRunCancelRequest {
|
|
3106
|
+
reason?: string;
|
|
3107
|
+
}
|
|
3108
|
+
export interface TrainingRunListParams {
|
|
3109
|
+
rule_id?: string;
|
|
3110
|
+
state?: TrainingRunState;
|
|
3111
|
+
limit?: number;
|
|
3112
|
+
offset?: number;
|
|
3113
|
+
}
|
|
3114
|
+
/** One firing, from the curated set to the decision. Money is frozen at fire time. */
|
|
3115
|
+
export interface TrainingRun {
|
|
3116
|
+
id: string;
|
|
3117
|
+
rule_id: string | null;
|
|
3118
|
+
rule_name: string | null;
|
|
3119
|
+
workspace_id: string;
|
|
3120
|
+
seq: number;
|
|
3121
|
+
rule_revision: number;
|
|
3122
|
+
authorized_by: string;
|
|
3123
|
+
rule_snapshot: Record<string, unknown>;
|
|
3124
|
+
trigger: TrainingTrigger;
|
|
3125
|
+
fired_reason: string | null;
|
|
3126
|
+
state: TrainingRunState;
|
|
3127
|
+
state_entered_at: string;
|
|
3128
|
+
attempts: number;
|
|
3129
|
+
next_poll_at: string | null;
|
|
3130
|
+
last_reason: string | null;
|
|
3131
|
+
error_code: string | null;
|
|
3132
|
+
last_error: string | null;
|
|
3133
|
+
training_ceiling_cents: number;
|
|
3134
|
+
candidate_ceiling_cents: number;
|
|
3135
|
+
eval_ceiling_cents: number;
|
|
3136
|
+
billed_training_cents: number;
|
|
3137
|
+
billed_candidate_cents: number;
|
|
3138
|
+
spent_eval_cents: number;
|
|
3139
|
+
loop_dataset_id: string | null;
|
|
3140
|
+
train_rows: number | null;
|
|
3141
|
+
holdout_rows: number | null;
|
|
3142
|
+
train_dataset_id: string | null;
|
|
3143
|
+
holdout_dataset_id: string | null;
|
|
3144
|
+
training_job_id: string | null;
|
|
3145
|
+
job_status: string | null;
|
|
3146
|
+
queue_deadline_at: string | null;
|
|
3147
|
+
checkpoint_id: string | null;
|
|
3148
|
+
checkpoint_step: number | null;
|
|
3149
|
+
training_eval_loss: number | null;
|
|
3150
|
+
candidate_deployment_id: string | null;
|
|
3151
|
+
candidate_name: string | null;
|
|
3152
|
+
candidate_seq: number;
|
|
3153
|
+
candidate_status: string | null;
|
|
3154
|
+
candidate_deadline_at: string | null;
|
|
3155
|
+
candidate_cleanup_at: string | null;
|
|
3156
|
+
evaluation_id: string | null;
|
|
3157
|
+
alias_name: string | null;
|
|
3158
|
+
alias_written_at: string | null;
|
|
3159
|
+
serving_before: TrainingRuleServing | null;
|
|
3160
|
+
rollback_available_until: string | null;
|
|
3161
|
+
verdict: TrainingVerdict | null;
|
|
3162
|
+
decision: TrainingDecision | null;
|
|
3163
|
+
decided_by: string | null;
|
|
3164
|
+
decided_at: string | null;
|
|
3165
|
+
auto_promote_at: string | null;
|
|
3166
|
+
review_deadline_at: string | null;
|
|
3167
|
+
notify_state: NotifyState;
|
|
3168
|
+
notify_event: string | null;
|
|
3169
|
+
notify_error: string | null;
|
|
3170
|
+
notify_attempts: number;
|
|
3171
|
+
created_at: string;
|
|
3172
|
+
updated_at: string;
|
|
3173
|
+
finished_at: string | null;
|
|
3174
|
+
}
|
|
3175
|
+
/** One row of the Runs table, carrying everything the list draws. */
|
|
3176
|
+
export interface TrainingRunSummary {
|
|
3177
|
+
id: string;
|
|
3178
|
+
rule_id: string | null;
|
|
3179
|
+
rule_name: string | null;
|
|
3180
|
+
workspace_id: string;
|
|
3181
|
+
seq: number;
|
|
3182
|
+
trigger: TrainingTrigger;
|
|
3183
|
+
state: TrainingRunState;
|
|
3184
|
+
state_entered_at: string;
|
|
3185
|
+
last_reason: string | null;
|
|
3186
|
+
error_code: string | null;
|
|
3187
|
+
verdict: TrainingVerdict | null;
|
|
3188
|
+
decision: TrainingDecision | null;
|
|
3189
|
+
train_rows: number | null;
|
|
3190
|
+
holdout_rows: number | null;
|
|
3191
|
+
billed_training_cents: number;
|
|
3192
|
+
billed_candidate_cents: number;
|
|
3193
|
+
spent_eval_cents: number;
|
|
3194
|
+
training_job_id: string | null;
|
|
3195
|
+
candidate_deployment_id: string | null;
|
|
3196
|
+
evaluation_id: string | null;
|
|
3197
|
+
review_deadline_at: string | null;
|
|
3198
|
+
auto_promote_at: string | null;
|
|
3199
|
+
notify_state: NotifyState;
|
|
3200
|
+
created_at: string;
|
|
3201
|
+
finished_at: string | null;
|
|
3202
|
+
}
|
|
3203
|
+
/** One line of the timeline. detail holds codes, ids and cents, never prompt text. */
|
|
3204
|
+
export interface TrainingRunEvent {
|
|
3205
|
+
seq: number;
|
|
3206
|
+
ts: string;
|
|
3207
|
+
from_state: TrainingRunState | null;
|
|
3208
|
+
to_state: TrainingRunState;
|
|
3209
|
+
reason: string;
|
|
3210
|
+
error_code: string | null;
|
|
3211
|
+
detail: Record<string, unknown>;
|
|
3212
|
+
actor: string;
|
|
3213
|
+
}
|
|
3214
|
+
export interface TrainingRunLinks {
|
|
3215
|
+
training_job_url: string | null;
|
|
3216
|
+
candidate_url: string | null;
|
|
3217
|
+
dataset_url: string | null;
|
|
3218
|
+
evaluation_url: string | null;
|
|
3219
|
+
}
|
|
3220
|
+
/** What this reader may do right now. A button that cannot work is never shown. */
|
|
3221
|
+
export interface TrainingRunActions {
|
|
3222
|
+
promote: boolean;
|
|
3223
|
+
reject: boolean;
|
|
3224
|
+
rollback: boolean;
|
|
3225
|
+
cancel: boolean;
|
|
3226
|
+
}
|
|
3227
|
+
export interface EvaluationRef {
|
|
3228
|
+
kind: ServingKind;
|
|
3229
|
+
name: string;
|
|
3230
|
+
deployment_id: string | null;
|
|
3231
|
+
model_version: string | null;
|
|
3232
|
+
}
|
|
3233
|
+
/** Frozen on the evaluation. Both sides are regenerated with identical decoding. */
|
|
3234
|
+
export interface EvaluationDecoding {
|
|
3235
|
+
temperature: number;
|
|
3236
|
+
max_tokens: number;
|
|
3237
|
+
stream: boolean;
|
|
3238
|
+
}
|
|
3239
|
+
export interface EvaluationJudgeDimension {
|
|
3240
|
+
key: string;
|
|
3241
|
+
description: string;
|
|
3242
|
+
}
|
|
3243
|
+
export interface EvaluationDimensionScore {
|
|
3244
|
+
dimension: string;
|
|
3245
|
+
incumbent: number;
|
|
3246
|
+
candidate: number;
|
|
3247
|
+
delta: number;
|
|
3248
|
+
}
|
|
3249
|
+
export interface EvaluationGraderScore {
|
|
3250
|
+
grader_id: string;
|
|
3251
|
+
name: string;
|
|
3252
|
+
incumbent: number;
|
|
3253
|
+
candidate: number;
|
|
3254
|
+
delta: number;
|
|
3255
|
+
/** A grader that applied to four rows has not measured anything. */
|
|
3256
|
+
items_applied: number;
|
|
3257
|
+
}
|
|
3258
|
+
export interface EvaluationWarning {
|
|
3259
|
+
code: EvaluationWarningCode;
|
|
3260
|
+
message: string;
|
|
3261
|
+
}
|
|
3262
|
+
/** The thresholds this verdict was measured against, frozen with the report. */
|
|
3263
|
+
export interface EvaluationMargin {
|
|
3264
|
+
promote_margin: number;
|
|
3265
|
+
promote_min_win_rate: number;
|
|
3266
|
+
min_judge_agreement: number;
|
|
3267
|
+
min_holdout_rows: number;
|
|
3268
|
+
}
|
|
3269
|
+
/**
|
|
3270
|
+
* One rubric dimension of a judge-versus-human agreement.
|
|
3271
|
+
*
|
|
3272
|
+
* Deliberately not EvaluationDimensionScore: that type carries `incumbent` and
|
|
3273
|
+
* `candidate`, which are the two models being compared, and an agreement has
|
|
3274
|
+
* neither. `pairs` is per dimension because a reviewer who rated one dimension
|
|
3275
|
+
* and skipped another leaves a different denominator behind each number.
|
|
3276
|
+
*/
|
|
3277
|
+
export interface JudgeAgreementDimension {
|
|
3278
|
+
dimension: string;
|
|
3279
|
+
pairs: number;
|
|
3280
|
+
agreement: number | null;
|
|
3281
|
+
mean_abs_error: number | null;
|
|
3282
|
+
}
|
|
3283
|
+
/** How often this judge agreed with the workspace's own reviewers. */
|
|
3284
|
+
export interface JudgeAgreement {
|
|
3285
|
+
judge_id: string;
|
|
3286
|
+
pairs: number;
|
|
3287
|
+
agreement: number | null;
|
|
3288
|
+
mean_abs_error: number | null;
|
|
3289
|
+
per_dimension: JudgeAgreementDimension[];
|
|
3290
|
+
window_days: number;
|
|
3291
|
+
computed_at: string;
|
|
3292
|
+
/** "Not enough reviewer overlap yet" is an answer; 100% of two is not. */
|
|
3293
|
+
enough_pairs: boolean;
|
|
3294
|
+
}
|
|
3295
|
+
/** The comparison report: the candidate against what serves today, same rows. */
|
|
3296
|
+
export interface Evaluation {
|
|
3297
|
+
id: string;
|
|
3298
|
+
run_id: string;
|
|
3299
|
+
workspace_id: string;
|
|
3300
|
+
loop_dataset_id: string | null;
|
|
3301
|
+
split: string;
|
|
3302
|
+
incumbent_ref: EvaluationRef;
|
|
3303
|
+
candidate_ref: EvaluationRef;
|
|
3304
|
+
judge_id: string | null;
|
|
3305
|
+
judge_name: string | null;
|
|
3306
|
+
judge_model: string | null;
|
|
3307
|
+
judge_instructions: string | null;
|
|
3308
|
+
judge_dimensions: EvaluationJudgeDimension[];
|
|
3309
|
+
judge_system_prompt: string | null;
|
|
3310
|
+
grader_ids: string[];
|
|
3311
|
+
decoding: EvaluationDecoding;
|
|
3312
|
+
rows_selected: number;
|
|
3313
|
+
rows_scored: number;
|
|
3314
|
+
rows_failed: number;
|
|
3315
|
+
/** win_rate is wins / (wins + losses). Ties are excluded and reported separately. */
|
|
3316
|
+
wins: number | null;
|
|
3317
|
+
losses: number | null;
|
|
3318
|
+
ties: number | null;
|
|
3319
|
+
win_rate: number | null;
|
|
3320
|
+
incumbent_mean: number | null;
|
|
3321
|
+
candidate_mean: number | null;
|
|
3322
|
+
mean_delta: number | null;
|
|
3323
|
+
judge_incumbent_mean: number | null;
|
|
3324
|
+
judge_candidate_mean: number | null;
|
|
3325
|
+
grader_incumbent_mean: number | null;
|
|
3326
|
+
grader_candidate_mean: number | null;
|
|
3327
|
+
grader_items_applied: number;
|
|
3328
|
+
per_dimension: EvaluationDimensionScore[];
|
|
3329
|
+
per_grader: EvaluationGraderScore[];
|
|
3330
|
+
judge_agreement: JudgeAgreement | null;
|
|
3331
|
+
/** Informational: there is no incumbent counterpart to compare it against. */
|
|
3332
|
+
trainer_eval_loss: number | null;
|
|
3333
|
+
warnings: EvaluationWarning[];
|
|
3334
|
+
verdict: TrainingVerdict | null;
|
|
3335
|
+
verdict_reason: string | null;
|
|
3336
|
+
margin_used: EvaluationMargin | null;
|
|
3337
|
+
prompt_tokens: number;
|
|
3338
|
+
completion_tokens: number;
|
|
3339
|
+
spent_cents: number;
|
|
3340
|
+
status: EvaluationStatus;
|
|
3341
|
+
last_error: string | null;
|
|
3342
|
+
created_at: string;
|
|
3343
|
+
finished_at: string | null;
|
|
3344
|
+
}
|
|
3345
|
+
/** One model's answer to one held-out conversation, with the scores it earned. */
|
|
3346
|
+
export interface EvaluationItemSide {
|
|
3347
|
+
completion: string | null;
|
|
3348
|
+
tool_calls: TrainingToolCall[] | null;
|
|
3349
|
+
grader: Record<string, unknown> | null;
|
|
3350
|
+
judge: Record<string, unknown> | null;
|
|
3351
|
+
score: number | null;
|
|
3352
|
+
model_version: string | null;
|
|
3353
|
+
}
|
|
3354
|
+
/**
|
|
3355
|
+
* One paired conversation behind the numbers.
|
|
3356
|
+
*
|
|
3357
|
+
* trace_id is nullable on purpose: deleting one conversation must not shrink a
|
|
3358
|
+
* finished report so that its stated n and its visible rows disagree.
|
|
3359
|
+
*/
|
|
3360
|
+
export interface EvaluationItem {
|
|
3361
|
+
id: number;
|
|
3362
|
+
dataset_item_id: number;
|
|
3363
|
+
trace_id: string | null;
|
|
3364
|
+
trace_url: string | null;
|
|
3365
|
+
prompt: TrainingMessage[];
|
|
3366
|
+
tools: unknown[] | null;
|
|
3367
|
+
incumbent: EvaluationItemSide;
|
|
3368
|
+
candidate: EvaluationItemSide;
|
|
3369
|
+
human_verdict: number | null;
|
|
3370
|
+
winner: EvaluationWinner | null;
|
|
3371
|
+
status: EvaluationItemStatus;
|
|
3372
|
+
error: string | null;
|
|
3373
|
+
}
|
|
3374
|
+
export interface EvaluationItemListParams {
|
|
3375
|
+
winner?: EvaluationWinner;
|
|
3376
|
+
/** Capped at 100 by the service. */
|
|
3377
|
+
limit?: number;
|
|
3378
|
+
offset?: number;
|
|
3379
|
+
}
|
|
3380
|
+
export interface JudgeAgreementParams {
|
|
3381
|
+
/** RFC 3339. Narrows the window the agreement is computed over. */
|
|
3382
|
+
from?: string;
|
|
3383
|
+
to?: string;
|
|
3384
|
+
}
|
|
3385
|
+
/** Which model, whose words, how much. The cap is pushed to the workspace key. */
|
|
3386
|
+
export interface AgentSettings {
|
|
3387
|
+
workspace_id: string;
|
|
3388
|
+
default_model: string | null;
|
|
3389
|
+
judge_system_prompt: string | null;
|
|
3390
|
+
sampler_system_prompt: string | null;
|
|
3391
|
+
eval_monthly_cap_cents: number | null;
|
|
3392
|
+
updated_by: string | null;
|
|
3393
|
+
updated_at: string;
|
|
3394
|
+
}
|
|
3395
|
+
/** Absent leaves a setting alone; a present null returns it to the platform default. */
|
|
3396
|
+
export interface AgentSettingsRequest {
|
|
3397
|
+
default_model?: string | null;
|
|
3398
|
+
judge_system_prompt?: string | null;
|
|
3399
|
+
sampler_system_prompt?: string | null;
|
|
3400
|
+
eval_monthly_cap_cents?: number | null;
|
|
3401
|
+
}
|
|
3402
|
+
/** A re-pointable public handle. Promotion is one row write here. */
|
|
3403
|
+
export interface InferenceAlias {
|
|
3404
|
+
workspace_id: string;
|
|
3405
|
+
name: string;
|
|
3406
|
+
target_inference_id: string;
|
|
3407
|
+
previous_target_inference_id: string | null;
|
|
3408
|
+
/** Set only at adoption, when the alias takes over a deployment's own name. */
|
|
3409
|
+
shadows_inference_id: string | null;
|
|
3410
|
+
set_by: string;
|
|
3411
|
+
origin: string | null;
|
|
3412
|
+
created_at: string;
|
|
3413
|
+
updated_at: string;
|
|
3414
|
+
}
|
|
3415
|
+
export interface InferenceAliasRequest {
|
|
3416
|
+
target_inference_id: string;
|
|
3417
|
+
origin?: string;
|
|
3418
|
+
}
|
|
3419
|
+
export interface TrainingRuleListResponse {
|
|
3420
|
+
rules: TrainingRule[];
|
|
3421
|
+
total: number;
|
|
3422
|
+
}
|
|
3423
|
+
export interface TrainingRuleResponse {
|
|
3424
|
+
rule: TrainingRule;
|
|
3425
|
+
recent_runs: TrainingRunSummary[];
|
|
3426
|
+
month_spent_cents: number;
|
|
3427
|
+
judge_agreement: JudgeAgreement | null;
|
|
3428
|
+
}
|
|
3429
|
+
export interface TrainingRuleMutationResponse {
|
|
3430
|
+
rule: TrainingRule;
|
|
3431
|
+
/** The rule will not fire again until someone confirms the new amounts. */
|
|
3432
|
+
consent_required: boolean;
|
|
3433
|
+
preflight: TrainingRulePreflight | null;
|
|
3434
|
+
}
|
|
3435
|
+
export interface TrainingRuleDeleteResponse {
|
|
3436
|
+
deleted: boolean;
|
|
3437
|
+
rule_id: string;
|
|
3438
|
+
cancelled_run_id: string | null;
|
|
3439
|
+
}
|
|
3440
|
+
export interface TrainingRunListResponse {
|
|
3441
|
+
runs: TrainingRunSummary[];
|
|
3442
|
+
total: number;
|
|
3443
|
+
}
|
|
3444
|
+
export interface TrainingRunResponse {
|
|
3445
|
+
run: TrainingRun;
|
|
3446
|
+
timeline: TrainingRunEvent[];
|
|
3447
|
+
links: TrainingRunLinks;
|
|
3448
|
+
available_actions: TrainingRunActions;
|
|
3449
|
+
}
|
|
3450
|
+
export interface TrainingRunActionResponse {
|
|
3451
|
+
run: TrainingRun;
|
|
3452
|
+
/** Set on rollback, the one action that changes what answers the traffic. */
|
|
3453
|
+
serving: TrainingRuleServing | null;
|
|
3454
|
+
}
|
|
3455
|
+
export interface EvaluationItemsResponse {
|
|
3456
|
+
items: EvaluationItem[];
|
|
3457
|
+
total: number;
|
|
3458
|
+
}
|
|
3459
|
+
export interface AgentSettingsResponse {
|
|
3460
|
+
settings: AgentSettings;
|
|
3461
|
+
}
|
|
3462
|
+
export interface InferenceAliasListResponse {
|
|
3463
|
+
aliases: InferenceAlias[];
|
|
3464
|
+
}
|
|
3465
|
+
export interface InferenceAliasResponse {
|
|
3466
|
+
alias: InferenceAlias;
|
|
3467
|
+
}
|
|
3468
|
+
export interface InferenceAliasDeleteResponse {
|
|
3469
|
+
deleted: boolean;
|
|
3470
|
+
}
|
package/dist/types.js
CHANGED
|
@@ -1,4 +1,14 @@
|
|
|
1
1
|
// ============================================================================
|
|
2
2
|
// Run BiOS SDK — Type Definitions
|
|
3
3
|
// ============================================================================
|
|
4
|
-
|
|
4
|
+
/** The closed set a run never leaves. */
|
|
5
|
+
export const TERMINAL_RUN_STATES = [
|
|
6
|
+
'promoted',
|
|
7
|
+
'rejected',
|
|
8
|
+
'expired',
|
|
9
|
+
'failed',
|
|
10
|
+
'budget_stopped',
|
|
11
|
+
'cancelled',
|
|
12
|
+
'superseded',
|
|
13
|
+
'rolled_back',
|
|
14
|
+
];
|