runbios-sdk 0.2.17 → 0.2.18-dev.271
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/dist/client.d.ts +10 -6
- package/dist/client.js +11 -7
- package/dist/index.d.ts +2 -2
- package/dist/index.js +1 -1
- package/dist/resources/datasets.d.ts +1 -25
- package/dist/resources/datasets.js +0 -39
- package/dist/resources/inference.js +1 -1
- package/dist/resources/loop.d.ts +150 -543
- package/dist/resources/loop.js +204 -723
- package/dist/types.d.ts +620 -1070
- package/dist/types.js +1 -1
- package/package.json +1 -1
package/dist/types.d.ts
CHANGED
|
@@ -8,13 +8,13 @@ export interface BiOSConfig {
|
|
|
8
8
|
orgId?: string;
|
|
9
9
|
/** Workspace ID. Optional when using API keys (resolved from the key). Can override for multi-workspace keys. */
|
|
10
10
|
workspaceId?: string;
|
|
11
|
-
/** Base URL for the API. Falls back to the RUNBIOS_BASE_URL environment variable (legacy: BIOS_BASE_URL), then the canonical https://api.runbios.ai hostname. */
|
|
11
|
+
/** Base URL for the API. Falls back to the RUNBIOS_BASE_URL environment variable (legacy: BIOS_BASE_URL), then the canonical https://api-dev.runbios.ai hostname. */
|
|
12
12
|
baseUrl?: string;
|
|
13
13
|
/** Request timeout in milliseconds. Defaults to 30000. */
|
|
14
14
|
timeout?: number;
|
|
15
15
|
/** Default per-deployment inference key. Can be overridden per inference call. */
|
|
16
16
|
inferenceKey?: string;
|
|
17
|
-
/** Inference base URL. Defaults to baseUrl, then https://api.runbios.ai. */
|
|
17
|
+
/** Inference base URL. Defaults to baseUrl, then https://api-dev.runbios.ai. */
|
|
18
18
|
inferenceBaseUrl?: string;
|
|
19
19
|
/** End-to-end inference timeout in milliseconds. Defaults to 15 minutes. */
|
|
20
20
|
inferenceTimeout?: number;
|
|
@@ -461,21 +461,6 @@ export interface DatasetRegisterHFParams {
|
|
|
461
461
|
columnMapping?: Record<string, DatasetTrainingField>;
|
|
462
462
|
importMode?: 'auto' | 'reference' | 'materialize';
|
|
463
463
|
}
|
|
464
|
-
/** Parameters for registering a curated Conscious Loop set as a dataset. */
|
|
465
|
-
export interface DatasetRegisterLoopParams {
|
|
466
|
-
/** The Conscious Loop dataset to register. */
|
|
467
|
-
loopDatasetId: string;
|
|
468
|
-
/** Defaults to `loop-<first 8 of the loop id>`. */
|
|
469
|
-
name?: string;
|
|
470
|
-
/**
|
|
471
|
-
* Which rows to take. Defaults to `all`, because the loop already split the
|
|
472
|
-
* set by TIME and re-splitting a deliberate split throws that decision away.
|
|
473
|
-
* Register `train` and `holdout` separately to score a training job against
|
|
474
|
-
* the loop's own held-back rows; both can exist at once.
|
|
475
|
-
*/
|
|
476
|
-
split?: 'all' | 'train' | 'holdout';
|
|
477
|
-
workspaceId?: string;
|
|
478
|
-
}
|
|
479
464
|
/** Parameters for searching HuggingFace Hub datasets. */
|
|
480
465
|
export interface DatasetHubSearchParams {
|
|
481
466
|
/** Search query. */
|
|
@@ -2344,7 +2329,12 @@ export interface LoopMessage {
|
|
|
2344
2329
|
name?: string;
|
|
2345
2330
|
}
|
|
2346
2331
|
export interface LoopCaptureParams {
|
|
2347
|
-
/**
|
|
2332
|
+
/**
|
|
2333
|
+
* The source being recorded. Must be switched on, or nothing is stored. A
|
|
2334
|
+
* pipeline's own tag (`pipeline.own_tag`) is a source its create switched
|
|
2335
|
+
* on, stamping that tag, so sending there reaches the pipeline with no
|
|
2336
|
+
* further setup; any other source is switched on with `setConfig`.
|
|
2337
|
+
*/
|
|
2348
2338
|
deployment_id: string;
|
|
2349
2339
|
model: string;
|
|
2350
2340
|
messages: LoopMessage[];
|
|
@@ -2362,6 +2352,12 @@ export interface LoopCaptureParams {
|
|
|
2362
2352
|
completion_tokens?: number;
|
|
2363
2353
|
latency_ms?: number;
|
|
2364
2354
|
metadata?: Record<string, unknown>;
|
|
2355
|
+
/**
|
|
2356
|
+
* Tags for this conversation, on top of the ones its source stamps. A
|
|
2357
|
+
* pipeline trains on every sample carrying any of its tags, so this is how
|
|
2358
|
+
* one conversation reaches another pipeline too. At most 32 in all.
|
|
2359
|
+
*/
|
|
2360
|
+
labels?: string[];
|
|
2365
2361
|
}
|
|
2366
2362
|
export interface LoopCaptureResult {
|
|
2367
2363
|
captured: boolean;
|
|
@@ -2369,7 +2365,6 @@ export interface LoopCaptureResult {
|
|
|
2369
2365
|
trace_id?: string;
|
|
2370
2366
|
/** Set when something was removed before the record was written. */
|
|
2371
2367
|
redacted?: boolean;
|
|
2372
|
-
download_url?: string;
|
|
2373
2368
|
/** Why nothing was stored, e.g. `capture_disabled`, `not_sampled`. */
|
|
2374
2369
|
reason?: string;
|
|
2375
2370
|
}
|
|
@@ -2554,8 +2549,8 @@ export interface LoopSignalParams {
|
|
|
2554
2549
|
/**
|
|
2555
2550
|
* Which training pipeline this feedback feeds, named in the SAME call.
|
|
2556
2551
|
*
|
|
2557
|
-
* A
|
|
2558
|
-
*
|
|
2552
|
+
* A pipeline trains on the conversations carrying its tags, so a label IS
|
|
2553
|
+
* the pipeline a conversation goes down.
|
|
2559
2554
|
* Putting one on used to need a second request after this one — and that
|
|
2560
2555
|
* second request is the one that gets skipped, by a script that handles the
|
|
2561
2556
|
* 201 and moves on, by a retry that succeeds here and fails there, by an
|
|
@@ -2622,6 +2617,18 @@ export interface LoopTraceListParams {
|
|
|
2622
2617
|
model?: string;
|
|
2623
2618
|
/** `captured` (our model's own behaviour) or `imported` (a file somebody brought). */
|
|
2624
2619
|
origin?: 'captured' | 'imported';
|
|
2620
|
+
/**
|
|
2621
|
+
* Only the conversations one pipeline's selection covers -- any of its tags
|
|
2622
|
+
* -- reviewed or not. The other filters narrow it further.
|
|
2623
|
+
*/
|
|
2624
|
+
pipeline?: string;
|
|
2625
|
+
/**
|
|
2626
|
+
* Only the conversations whose deciding verdict is this feedback: `good`
|
|
2627
|
+
* (accepted), `bad` (rejected) or `corrected` (edited, or gold).
|
|
2628
|
+
*/
|
|
2629
|
+
signal?: 'good' | 'bad' | 'corrected';
|
|
2630
|
+
/** Only the conversations nobody has given feedback on yet. */
|
|
2631
|
+
unsignalled?: boolean;
|
|
2625
2632
|
}
|
|
2626
2633
|
export interface LoopTraceListResponse {
|
|
2627
2634
|
traces: LoopTrace[];
|
|
@@ -2629,69 +2636,6 @@ export interface LoopTraceListResponse {
|
|
|
2629
2636
|
limit: number;
|
|
2630
2637
|
offset: number;
|
|
2631
2638
|
}
|
|
2632
|
-
export interface LoopDatasetCreateParams {
|
|
2633
|
-
name: string;
|
|
2634
|
-
method: LoopMethod;
|
|
2635
|
-
deployment_id?: string;
|
|
2636
|
-
from?: string;
|
|
2637
|
-
to?: string;
|
|
2638
|
-
/** Narrow to one slice. A label with children selects them too. */
|
|
2639
|
-
label?: string;
|
|
2640
|
-
/**
|
|
2641
|
-
* Take this many of what the filters matched. Which ones is arbitrary but
|
|
2642
|
-
* REPEATABLE: the same conversations always give the same slice, so two
|
|
2643
|
-
* builds of one spec describe the same set.
|
|
2644
|
-
*/
|
|
2645
|
-
sample?: number;
|
|
2646
|
-
/** Holds back your most recent work, not a random slice. 0-50. */
|
|
2647
|
-
holdout_percent?: number;
|
|
2648
|
-
max_items?: number;
|
|
2649
|
-
/**
|
|
2650
|
-
* Named dimensions, ANDed with each other and with `label`:
|
|
2651
|
-
* `{ category: 'billing', language: 'es' }`. This is what turns one captured
|
|
2652
|
-
* corpus into a different dataset for every task somebody trains for.
|
|
2653
|
-
*/
|
|
2654
|
-
attributes?: Record<string, string>;
|
|
2655
|
-
/** Only the conversations nobody has described yet. */
|
|
2656
|
-
unlabelled?: boolean;
|
|
2657
|
-
/** Only what ONE model answered — the filter distillation is made of. */
|
|
2658
|
-
model?: string;
|
|
2659
|
-
/** `captured` (our model's own behaviour) or `imported` (a file somebody brought). */
|
|
2660
|
-
origin?: 'captured' | 'imported';
|
|
2661
|
-
}
|
|
2662
|
-
export interface LoopDataset {
|
|
2663
|
-
id: string;
|
|
2664
|
-
workspace_id: string;
|
|
2665
|
-
name: string;
|
|
2666
|
-
method: LoopMethod;
|
|
2667
|
-
status: string;
|
|
2668
|
-
spec: Record<string, unknown>;
|
|
2669
|
-
item_count: number;
|
|
2670
|
-
considered_count: number;
|
|
2671
|
-
/** Why rows were left out, by reason. */
|
|
2672
|
-
rejected_counts: Record<string, number>;
|
|
2673
|
-
holdout_count: number;
|
|
2674
|
-
holdout_cutoff: string | null;
|
|
2675
|
-
created_by: string | null;
|
|
2676
|
-
created_at: string;
|
|
2677
|
-
completed_at: string | null;
|
|
2678
|
-
download_url: string;
|
|
2679
|
-
}
|
|
2680
|
-
export interface LoopDatasetListParams {
|
|
2681
|
-
method?: LoopMethod;
|
|
2682
|
-
limit?: number;
|
|
2683
|
-
offset?: number;
|
|
2684
|
-
}
|
|
2685
|
-
export interface LoopDatasetItem {
|
|
2686
|
-
id: number;
|
|
2687
|
-
trace_id: string;
|
|
2688
|
-
split: 'train' | 'holdout';
|
|
2689
|
-
payload: Record<string, unknown>;
|
|
2690
|
-
dedup_key: string;
|
|
2691
|
-
created_at: string;
|
|
2692
|
-
/** The conversation this row was built from. */
|
|
2693
|
-
trace_url: string;
|
|
2694
|
-
}
|
|
2695
2639
|
export interface LoopConfig {
|
|
2696
2640
|
deployment_id: string;
|
|
2697
2641
|
workspace_id: string;
|
|
@@ -2701,42 +2645,8 @@ export interface LoopConfig {
|
|
|
2701
2645
|
enabled_by: string | null;
|
|
2702
2646
|
enabled_at: string | null;
|
|
2703
2647
|
updated_at: string;
|
|
2704
|
-
|
|
2705
|
-
|
|
2706
|
-
export interface LoopBuildRuleParams {
|
|
2707
|
-
name: string;
|
|
2708
|
-
method: string;
|
|
2709
|
-
/**
|
|
2710
|
-
* How many rows reviewed SINCE THE LAST BUILD must exist before this fires
|
|
2711
|
-
* again. Minimum 100, and 100 when absent. Counting the whole corpus instead
|
|
2712
|
-
* would fire the rule every interval forever, because a total that has
|
|
2713
|
-
* crossed a threshold stays across it.
|
|
2714
|
-
*/
|
|
2715
|
-
min_new_rows?: number;
|
|
2716
|
-
enabled?: boolean;
|
|
2717
|
-
/** The same selection `createDataset` takes, replayed verbatim. */
|
|
2718
|
-
spec?: LoopDatasetCreateParams;
|
|
2719
|
-
}
|
|
2720
|
-
export interface LoopBuildRule {
|
|
2721
|
-
id: string;
|
|
2722
|
-
workspace_id: string;
|
|
2723
|
-
name: string;
|
|
2724
|
-
method: string;
|
|
2725
|
-
spec: Record<string, unknown>;
|
|
2726
|
-
min_new_rows: number;
|
|
2727
|
-
enabled: boolean;
|
|
2728
|
-
last_built_at: string | null;
|
|
2729
|
-
last_dataset_id: string | null;
|
|
2730
|
-
/**
|
|
2731
|
-
* Why the rule did not fire last time it was checked. A rule quiet because
|
|
2732
|
-
* it is waiting looks exactly like one quiet because it is broken; this is
|
|
2733
|
-
* the difference.
|
|
2734
|
-
*/
|
|
2735
|
-
last_reason: string | null;
|
|
2736
|
-
last_checked_at: string | null;
|
|
2737
|
-
created_by: string | null;
|
|
2738
|
-
created_at: string;
|
|
2739
|
-
updated_at: string;
|
|
2648
|
+
/** Tags stamped, server side, on every conversation captured from this source. */
|
|
2649
|
+
tags: string[];
|
|
2740
2650
|
}
|
|
2741
2651
|
export interface LoopConfigParams {
|
|
2742
2652
|
enabled: boolean;
|
|
@@ -2744,365 +2654,109 @@ export interface LoopConfigParams {
|
|
|
2744
2654
|
/** Deterministic per conversation, so turns are never split. 0 < rate <= 1. */
|
|
2745
2655
|
sample_rate?: number;
|
|
2746
2656
|
/**
|
|
2747
|
-
*
|
|
2748
|
-
*
|
|
2749
|
-
* that turned it on.
|
|
2750
|
-
*/
|
|
2751
|
-
auto_grade?: boolean;
|
|
2752
|
-
}
|
|
2753
|
-
/** One thing a judge scores, separately from the others. */
|
|
2754
|
-
export interface LoopJudgeDimension {
|
|
2755
|
-
key: string;
|
|
2756
|
-
description?: string;
|
|
2757
|
-
}
|
|
2758
|
-
/** Which slice of the corpus a judge is responsible for. */
|
|
2759
|
-
export interface LoopJudgeSelection {
|
|
2760
|
-
label?: string;
|
|
2761
|
-
deployment_id?: string;
|
|
2762
|
-
from?: string;
|
|
2763
|
-
to?: string;
|
|
2764
|
-
sample?: number;
|
|
2765
|
-
/** Skip conversations that already carry a judge verdict. */
|
|
2766
|
-
only_unscored?: boolean;
|
|
2767
|
-
}
|
|
2768
|
-
export interface LoopJudge {
|
|
2769
|
-
id: string;
|
|
2770
|
-
workspace_id: string;
|
|
2771
|
-
name: string;
|
|
2772
|
-
instructions: string;
|
|
2773
|
-
dimensions: LoopJudgeDimension[];
|
|
2774
|
-
selection: LoopJudgeSelection;
|
|
2775
|
-
model: string | null;
|
|
2776
|
-
write_gold: boolean;
|
|
2777
|
-
enabled: boolean;
|
|
2778
|
-
/** The platform's agent runs this judge, on the workspace's own serverless account. */
|
|
2779
|
-
auto: boolean;
|
|
2780
|
-
/**
|
|
2781
|
-
* Was automatic when the agent was turned off. Turning the agent back on
|
|
2782
|
-
* restores exactly these judges.
|
|
2783
|
-
*/
|
|
2784
|
-
auto_paused: boolean;
|
|
2785
|
-
/**
|
|
2786
|
-
* Why the last turn-on did NOT restore this judge. Null is the ordinary
|
|
2787
|
-
* state. Set when the resume declined to make it automatic because the model
|
|
2788
|
-
* it names is not one the serving gateway will route — handing it back would
|
|
2789
|
-
* buy a run that fails every conversation. It stays paused, so pointing it at
|
|
2790
|
-
* a model that routes and turning the agent on again brings it back.
|
|
2791
|
-
*/
|
|
2792
|
-
auto_pause_reason: string | null;
|
|
2793
|
-
created_by: string | null;
|
|
2794
|
-
created_at: string;
|
|
2795
|
-
updated_at: string;
|
|
2796
|
-
}
|
|
2797
|
-
export interface LoopJudgeParams {
|
|
2798
|
-
name: string;
|
|
2799
|
-
instructions: string;
|
|
2800
|
-
dimensions: LoopJudgeDimension[];
|
|
2801
|
-
selection?: LoopJudgeSelection;
|
|
2802
|
-
model?: string;
|
|
2803
|
-
/**
|
|
2804
|
-
* Let this judge write the answer that should have been given, not just a
|
|
2805
|
-
* score. Off by default: a reference answer written by a model and then
|
|
2806
|
-
* trained on is distillation, which is a decision you make deliberately.
|
|
2807
|
-
*/
|
|
2808
|
-
write_gold?: boolean;
|
|
2809
|
-
enabled?: boolean;
|
|
2810
|
-
/**
|
|
2811
|
-
* Hand the running of this judge to the platform's agent: new conversations
|
|
2812
|
-
* in its slice are scored as they arrive, one model call each, on THIS
|
|
2813
|
-
* WORKSPACE'S OWN serverless account (the agent spends through a managed key
|
|
2814
|
-
* of yours, "Conscious Loop" in your key list). Requires `model`. Off, you
|
|
2815
|
-
* run the model yourself with startRun / takeWork / postVerdicts.
|
|
2816
|
-
*/
|
|
2817
|
-
auto?: boolean;
|
|
2818
|
-
}
|
|
2819
|
-
/** One pass of one rubric over one slice. */
|
|
2820
|
-
export interface LoopJudgeRun {
|
|
2821
|
-
id: string;
|
|
2822
|
-
judge_id: string;
|
|
2823
|
-
judge_name?: string;
|
|
2824
|
-
/**
|
|
2825
|
-
* `abandoned` is a run you opened and never drained: nothing was scored on
|
|
2826
|
-
* it for a week and items were still waiting. Nothing is deleted and posting
|
|
2827
|
-
* verdicts to it still works and still closes it as done. It exists so that
|
|
2828
|
-
* `open` keeps meaning "somebody is working on this".
|
|
2829
|
-
*
|
|
2830
|
-
* `stopped` is a run somebody closed on purpose before it finished, with
|
|
2831
|
-
* `stopRun`. Separate from `done` because a run that covered three of forty
|
|
2832
|
-
* conversations did not finish its work: every score it recorded is kept
|
|
2833
|
-
* either way, and reading one as the other overstates what was evaluated.
|
|
2834
|
-
*/
|
|
2835
|
-
status: 'open' | 'done' | 'abandoned' | 'stopped';
|
|
2836
|
-
instructions: string;
|
|
2837
|
-
dimensions: LoopJudgeDimension[];
|
|
2838
|
-
model: string | null;
|
|
2839
|
-
selected: number;
|
|
2840
|
-
scored: number;
|
|
2841
|
-
failed: number;
|
|
2842
|
-
/** Who drains it: your own code ('caller') or the platform's agent ('platform'). */
|
|
2843
|
-
runner: 'caller' | 'platform';
|
|
2844
|
-
/** The agent's last complaint about this run, or null while it is working. */
|
|
2845
|
-
last_error: string | null;
|
|
2846
|
-
last_activity_at: string | null;
|
|
2847
|
-
created_by: string | null;
|
|
2848
|
-
created_at: string;
|
|
2849
|
-
finished_at: string | null;
|
|
2850
|
-
}
|
|
2851
|
-
/** The workspace's managed serverless key the agent spends through. Never the secret. */
|
|
2852
|
-
export interface LoopAgentCredential {
|
|
2853
|
-
workspace_id: string;
|
|
2854
|
-
key_id: string;
|
|
2855
|
-
key_prefix: string;
|
|
2856
|
-
created_by: string | null;
|
|
2857
|
-
created_at: string;
|
|
2858
|
-
updated_at: string;
|
|
2859
|
-
last_used_at: string | null;
|
|
2860
|
-
last_error: string | null;
|
|
2861
|
-
revoked_at: string | null;
|
|
2862
|
-
/**
|
|
2863
|
-
* The most the agent's key may spend on model calls in a calendar month,
|
|
2864
|
-
* in cents. `null` = no cap (the default): the agent can spend up to the
|
|
2865
|
-
* workspace's serverless balance. Once reached, the agent's model calls
|
|
2866
|
-
* are refused until next month and its runs pause with that reason.
|
|
2657
|
+
* Tags stamped on every conversation captured from this source, so live
|
|
2658
|
+
* traffic reaches the pipelines they feed. Omitted means unchanged; [] clears.
|
|
2867
2659
|
*/
|
|
2868
|
-
|
|
2869
|
-
}
|
|
2870
|
-
export interface LoopAgentStatus {
|
|
2871
|
-
/** An agent can exist in this environment at all. */
|
|
2872
|
-
available: boolean;
|
|
2873
|
-
/** Which piece is missing when it cannot. */
|
|
2874
|
-
reason: string;
|
|
2875
|
-
/** A worker has checked in within the last minute. */
|
|
2876
|
-
online: boolean;
|
|
2877
|
-
agent: {
|
|
2878
|
-
seen_at: string;
|
|
2879
|
-
passes: number;
|
|
2880
|
-
judge_items: number;
|
|
2881
|
-
samples: number;
|
|
2882
|
-
} | null;
|
|
2883
|
-
credential: LoopAgentCredential | null;
|
|
2884
|
-
open_judge_runs: number;
|
|
2885
|
-
open_sample_runs: number;
|
|
2886
|
-
auto_judges: number;
|
|
2887
|
-
}
|
|
2888
|
-
export interface LoopSampleSelection extends LoopJudgeSelection {
|
|
2889
|
-
/** Skip conversations that already have alternatives. */
|
|
2890
|
-
only_unsampled?: boolean;
|
|
2891
|
-
}
|
|
2892
|
-
export interface LoopSampleRunParams {
|
|
2893
|
-
/** The model that writes the alternatives. For distillation, the teacher. */
|
|
2894
|
-
model: string;
|
|
2895
|
-
/** Alternatives per conversation, 1-8. Default 4. */
|
|
2896
|
-
n?: number;
|
|
2897
|
-
/** 0-2. Default 0.8; 0 makes every sample the same answer. */
|
|
2898
|
-
temperature?: number;
|
|
2899
|
-
/** 16-8192. Default 1024. */
|
|
2900
|
-
max_tokens?: number;
|
|
2901
|
-
/** Score each sample with this judge as it is written. */
|
|
2902
|
-
judge_id?: string;
|
|
2903
|
-
selection?: LoopSampleSelection;
|
|
2904
|
-
}
|
|
2905
|
-
/** "Write N alternatives to each conversation in this slice, and score them." */
|
|
2906
|
-
export interface LoopSampleRun {
|
|
2907
|
-
id: string;
|
|
2908
|
-
workspace_id: string;
|
|
2909
|
-
model: string;
|
|
2910
|
-
n: number;
|
|
2911
|
-
temperature: number;
|
|
2912
|
-
max_tokens: number;
|
|
2913
|
-
judge_id: string | null;
|
|
2914
|
-
judge_name: string | null;
|
|
2915
|
-
selection: LoopSampleSelection;
|
|
2916
|
-
status: 'open' | 'done';
|
|
2917
|
-
selected: number;
|
|
2918
|
-
done: number;
|
|
2919
|
-
failed: number;
|
|
2920
|
-
samples: number;
|
|
2921
|
-
last_error: string | null;
|
|
2922
|
-
last_activity_at: string | null;
|
|
2923
|
-
created_by: string | null;
|
|
2924
|
-
created_at: string;
|
|
2925
|
-
finished_at: string | null;
|
|
2926
|
-
}
|
|
2927
|
-
/**
|
|
2928
|
-
* One conversation to score, already rendered into the prompt to send.
|
|
2929
|
-
*
|
|
2930
|
-
* Send `prompt` as-is. Assembling it yourself is how two callers end up giving
|
|
2931
|
-
* the same rubric different instructions, and two judges given different
|
|
2932
|
-
* instructions are not one judge.
|
|
2933
|
-
*/
|
|
2934
|
-
export interface LoopJudgeWorkItem {
|
|
2935
|
-
trace_id: string;
|
|
2936
|
-
messages: Array<{
|
|
2937
|
-
role: string;
|
|
2938
|
-
content?: string;
|
|
2939
|
-
}>;
|
|
2940
|
-
answer: string;
|
|
2941
|
-
prompt: string;
|
|
2942
|
-
}
|
|
2943
|
-
export interface LoopJudgeWork {
|
|
2944
|
-
run: LoopJudgeRun;
|
|
2945
|
-
items: LoopJudgeWorkItem[];
|
|
2946
|
-
remaining: number;
|
|
2947
|
-
}
|
|
2948
|
-
export type LoopJudgeRunItemStatus = 'pending' | 'scored' | 'failed';
|
|
2949
|
-
/** What happened to one conversation in one run. */
|
|
2950
|
-
export interface LoopJudgeRunItem {
|
|
2951
|
-
trace_id: string;
|
|
2952
|
-
status: LoopJudgeRunItemStatus;
|
|
2953
|
-
/**
|
|
2954
|
-
* Why this one could not be scored, as the caller reported it. This is where
|
|
2955
|
-
* a wrong model name or a refused key shows up; null for anything that is
|
|
2956
|
-
* not failed.
|
|
2957
|
-
*/
|
|
2958
|
-
error: string | null;
|
|
2959
|
-
scored_at: string | null;
|
|
2960
|
-
}
|
|
2961
|
-
export interface LoopJudgeRunItems {
|
|
2962
|
-
run: LoopJudgeRun;
|
|
2963
|
-
items: LoopJudgeRunItem[];
|
|
2964
|
-
/** True when the page ended before the run did. */
|
|
2965
|
-
has_more: boolean;
|
|
2966
|
-
/** Where to carry on from. Present only when `has_more`. */
|
|
2967
|
-
next_offset?: number;
|
|
2968
|
-
}
|
|
2969
|
-
/** One scored conversation going back. */
|
|
2970
|
-
export interface LoopJudgeVerdict {
|
|
2971
|
-
trace_id: string;
|
|
2972
|
-
/** One score per dimension the rubric asked for, each between 0 and 1. */
|
|
2973
|
-
scores?: Record<string, number>;
|
|
2974
|
-
/** Your own combination of them. Left out, the plain mean is used. */
|
|
2975
|
-
overall?: number;
|
|
2976
|
-
reason?: string;
|
|
2977
|
-
/** Only stored if the judge was created with `write_gold`. */
|
|
2978
|
-
gold?: string;
|
|
2979
|
-
/** Mark an item you could not score, instead of dropping it silently. */
|
|
2980
|
-
error?: string;
|
|
2981
|
-
}
|
|
2982
|
-
export interface LoopJudgeVerdictResult {
|
|
2983
|
-
run: LoopJudgeRun;
|
|
2984
|
-
recorded: number;
|
|
2985
|
-
failed: number;
|
|
2986
|
-
rejected: Array<{
|
|
2987
|
-
trace_id: string;
|
|
2988
|
-
error: string;
|
|
2989
|
-
}>;
|
|
2660
|
+
tags?: string[];
|
|
2990
2661
|
}
|
|
2662
|
+
/** GET /api/loop/stats: counts for the workspace. */
|
|
2991
2663
|
export interface LoopStats {
|
|
2992
2664
|
traces: number;
|
|
2993
2665
|
signalled_traces: number;
|
|
2994
2666
|
signals: number;
|
|
2995
2667
|
by_verdict: Record<string, number>;
|
|
2996
2668
|
by_source: Record<string, number>;
|
|
2997
|
-
datasets: number;
|
|
2998
|
-
/** Upper bounds: deduplication runs when a set is built. */
|
|
2999
|
-
ready: Record<LoopMethod, number>;
|
|
3000
|
-
}
|
|
3001
|
-
export interface LoopCandidateParams {
|
|
3002
|
-
completion?: string;
|
|
3003
|
-
/** An alternative that is itself a tool call. */
|
|
3004
|
-
tool_calls?: LoopToolCall[];
|
|
3005
|
-
/** Which model produced this alternative. The teacher, when distilling. */
|
|
3006
|
-
model?: string;
|
|
3007
|
-
model_version?: string;
|
|
3008
|
-
/** Unscored alternatives cannot pair -- a missing score is not a low one. */
|
|
3009
|
-
score?: number;
|
|
3010
|
-
/** Required alongside a score: human | verifier | judge | behavioural. */
|
|
3011
|
-
score_source?: LoopSource;
|
|
3012
|
-
reason?: string;
|
|
3013
|
-
metadata?: Record<string, unknown>;
|
|
3014
|
-
}
|
|
3015
|
-
export interface LoopCandidate {
|
|
3016
|
-
id: string;
|
|
3017
|
-
trace_id: string;
|
|
3018
|
-
model: string | null;
|
|
3019
|
-
model_version: string | null;
|
|
3020
|
-
completion: string;
|
|
3021
|
-
tool_calls?: LoopToolCall[];
|
|
3022
|
-
score: number | null;
|
|
3023
|
-
score_source: LoopSource | null;
|
|
3024
|
-
reason: string | null;
|
|
3025
|
-
metadata: Record<string, unknown>;
|
|
3026
|
-
created_at: string;
|
|
3027
|
-
}
|
|
3028
|
-
/**
|
|
3029
|
-
* The deterministic checks a grader can perform.
|
|
3030
|
-
*
|
|
3031
|
-
* `matches_gold` is the one kind with no expected value of its own: it
|
|
3032
|
-
* compares each answer to the gold answer recorded on that same conversation
|
|
3033
|
-
* (or the human correction when there is no gold), scores sampled
|
|
3034
|
-
* alternatives against the same gold, and is NOT APPLIED to a conversation
|
|
3035
|
-
* that carries neither, so it never marks down an unlabelled answer.
|
|
3036
|
-
*/
|
|
3037
|
-
export type LoopGraderKind = 'matches_gold' | 'exact_match' | 'contains' | 'regex' | 'json_valid' | 'numeric' | 'tool_called';
|
|
3038
|
-
export interface LoopGraderConfig {
|
|
3039
|
-
expected?: string;
|
|
3040
|
-
/** Every phrase that must appear. */
|
|
3041
|
-
required?: string[];
|
|
3042
|
-
/** Phrases that must not. On their own these are passed by silence. */
|
|
3043
|
-
forbidden?: string[];
|
|
3044
|
-
pattern?: string;
|
|
3045
|
-
/** Keys the answer must carry, for `json_valid`. */
|
|
3046
|
-
keys?: string[];
|
|
3047
2669
|
/**
|
|
3048
|
-
* How
|
|
3049
|
-
*
|
|
2670
|
+
* How many samples a pipeline could train on now: the ones marked good or
|
|
2671
|
+
* corrected. A pipeline trains SFT only, so this is the one count. An upper
|
|
2672
|
+
* bound: duplicates, the held-back set and conversations too long to train
|
|
2673
|
+
* on come out of it when an attempt trains.
|
|
3050
2674
|
*/
|
|
3051
|
-
|
|
3052
|
-
|
|
3053
|
-
|
|
3054
|
-
case_sensitive?: boolean;
|
|
2675
|
+
ready: {
|
|
2676
|
+
sft: number;
|
|
2677
|
+
};
|
|
3055
2678
|
}
|
|
3056
|
-
|
|
2679
|
+
/** One pipeline a conversation falls inside (GET /api/loop/traces/{id}/pipelines). */
|
|
2680
|
+
export interface LoopTracePipeline {
|
|
2681
|
+
/** The pipeline's id: `getPipeline(rule_id)`, the Pipeline's own `rule_id`. */
|
|
2682
|
+
rule_id: string;
|
|
2683
|
+
/** The pipeline's name. */
|
|
3057
2684
|
name: string;
|
|
3058
|
-
|
|
3059
|
-
config: LoopGraderConfig;
|
|
3060
|
-
/** The multiplier. Twice as important, twice the weight. Must exceed zero. */
|
|
3061
|
-
weight?: number;
|
|
3062
|
-
enabled?: boolean;
|
|
3063
|
-
/** Limit the rule to one capture source. Omit to apply it everywhere. */
|
|
3064
|
-
deployment_id?: string;
|
|
3065
|
-
/**
|
|
3066
|
-
* Aim the rule at one slice of the corpus. A master group covers its
|
|
3067
|
-
* children. Outside that slice the rule is NOT APPLIED, rather than failed,
|
|
3068
|
-
* so it stays out of the score entirely. Omit to apply it everywhere.
|
|
3069
|
-
*/
|
|
3070
|
-
label?: string;
|
|
3071
|
-
}
|
|
3072
|
-
export interface LoopGrader extends LoopGraderParams {
|
|
3073
|
-
id: string;
|
|
3074
|
-
workspace_id: string;
|
|
3075
|
-
weight: number;
|
|
2685
|
+
/** Whether the pipeline is switched on. A paused one still selects the conversation. */
|
|
3076
2686
|
enabled: boolean;
|
|
3077
|
-
|
|
2687
|
+
}
|
|
2688
|
+
export interface LoopTracePipelinesResponse {
|
|
2689
|
+
/** Live pipelines whose tags select this conversation. Empty is a real answer. */
|
|
2690
|
+
pipelines: LoopTracePipeline[];
|
|
2691
|
+
/** Being selected is not being trained on; this sentence says so. */
|
|
2692
|
+
note: string;
|
|
2693
|
+
}
|
|
2694
|
+
export interface LoopMetricsHistoryParams {
|
|
2695
|
+
/** One pipeline's id. Absent, every attempt in the workspace. */
|
|
2696
|
+
rule_id?: string;
|
|
2697
|
+
/** Most recent attempts to return: 200 when absent, at most 500. */
|
|
2698
|
+
limit?: number;
|
|
2699
|
+
/** Days of feedback counts: 90 when absent, at most 365. */
|
|
2700
|
+
days?: number;
|
|
2701
|
+
}
|
|
2702
|
+
/** One attempt, and the comparison it produced if it produced one. */
|
|
2703
|
+
export interface LoopMetricPoint {
|
|
2704
|
+
run_id: string;
|
|
2705
|
+
seq: number;
|
|
2706
|
+
rule_id: string | null;
|
|
2707
|
+
rule_name: string | null;
|
|
2708
|
+
state: TrainingRunState;
|
|
2709
|
+
verdict: TrainingVerdict | null;
|
|
2710
|
+
decision: TrainingDecision | null;
|
|
3078
2711
|
created_at: string;
|
|
3079
|
-
|
|
2712
|
+
finished_at: string | null;
|
|
2713
|
+
train_rows: number | null;
|
|
2714
|
+
holdout_rows: number | null;
|
|
2715
|
+
billed_training_cents: number;
|
|
2716
|
+
billed_candidate_cents: number;
|
|
2717
|
+
spent_eval_cents: number;
|
|
2718
|
+
benchmark_spent_cents: number;
|
|
2719
|
+
evaluation_id: string | null;
|
|
2720
|
+
rows_scored: number | null;
|
|
2721
|
+
rows_failed: number | null;
|
|
2722
|
+
wins: number | null;
|
|
2723
|
+
losses: number | null;
|
|
2724
|
+
ties: number | null;
|
|
2725
|
+
win_rate: number | null;
|
|
2726
|
+
incumbent_mean: number | null;
|
|
2727
|
+
candidate_mean: number | null;
|
|
2728
|
+
mean_delta: number | null;
|
|
2729
|
+
judge_incumbent_mean: number | null;
|
|
2730
|
+
judge_candidate_mean: number | null;
|
|
2731
|
+
trainer_eval_loss: number | null;
|
|
2732
|
+
/** The bar this attempt had to clear, frozen with its report. */
|
|
2733
|
+
promote_margin: number | null;
|
|
2734
|
+
promote_min_win_rate: number | null;
|
|
2735
|
+
checkpoint_id: string | null;
|
|
2736
|
+
checkpoint_step: number | null;
|
|
2737
|
+
attempt_no: number | null;
|
|
2738
|
+
/** The version this attempt became; null for every attempt that did not. */
|
|
2739
|
+
version_no: number | null;
|
|
2740
|
+
base_model_id: string | null;
|
|
2741
|
+
model_family: string | null;
|
|
2742
|
+
base_model_revision: string | null;
|
|
2743
|
+
train_type: string | null;
|
|
2744
|
+
trigger: string;
|
|
2745
|
+
recipe_note: string | null;
|
|
2746
|
+
recipe_of_version: number | null;
|
|
3080
2747
|
}
|
|
3081
|
-
|
|
3082
|
-
|
|
3083
|
-
|
|
3084
|
-
|
|
3085
|
-
|
|
3086
|
-
|
|
3087
|
-
|
|
3088
|
-
/**
|
|
3089
|
-
|
|
3090
|
-
|
|
3091
|
-
|
|
3092
|
-
|
|
3093
|
-
export interface LoopGradeReport {
|
|
3094
|
-
/** Weighted mean over the rules that applied, 0 to 1. */
|
|
3095
|
-
score: number;
|
|
3096
|
-
results: LoopGraderResult[];
|
|
3097
|
-
/** The denominator. 1.0 from one rule is not 1.0 from six. */
|
|
3098
|
-
applied: number;
|
|
3099
|
-
skipped: number;
|
|
3100
|
-
}
|
|
3101
|
-
export interface LoopGradeResult {
|
|
3102
|
-
report: LoopGradeReport;
|
|
3103
|
-
/** Null when no rule applied, because no verdict was invented. */
|
|
3104
|
-
signal_id: string | null;
|
|
3105
|
-
candidates_graded: number;
|
|
2748
|
+
/** One day of feedback, counted by verdict. The workspace's, whatever rule_id says. */
|
|
2749
|
+
export interface LoopSignalDay {
|
|
2750
|
+
day: string;
|
|
2751
|
+
total: number;
|
|
2752
|
+
by_verdict: Record<string, number>;
|
|
2753
|
+
}
|
|
2754
|
+
export interface LoopMetricsHistory {
|
|
2755
|
+
/** Oldest first, so a chart reads left to right. */
|
|
2756
|
+
runs: LoopMetricPoint[];
|
|
2757
|
+
signals: LoopSignalDay[];
|
|
2758
|
+
/** True when older attempts exist beyond `limit`. */
|
|
2759
|
+
truncated: boolean;
|
|
3106
2760
|
}
|
|
3107
2761
|
export interface LoopLabel {
|
|
3108
2762
|
label: string;
|
|
@@ -3144,8 +2798,6 @@ export interface LoopLabelCount {
|
|
|
3144
2798
|
traces: number;
|
|
3145
2799
|
key: string;
|
|
3146
2800
|
}
|
|
3147
|
-
export type TrainingCadence = 'none' | 'daily' | 'weekly';
|
|
3148
|
-
export type TrainingCombinator = 'and' | 'or';
|
|
3149
2801
|
/** What answers the customer's traffic today, and therefore what promotion re-points. */
|
|
3150
2802
|
export type ServingKind = 'serverless_slug' | 'deployment';
|
|
3151
2803
|
export type TrainingRulePausedReason = 'consent_invalid' | 'insufficient_funds' | 'authorizer_not_member' | 'model_not_trainable' | 'platform_capacity'
|
|
@@ -3157,45 +2809,6 @@ export type TrainingRulePausedReason = 'consent_invalid' | 'insufficient_funds'
|
|
|
3157
2809
|
* wrong; the platform team enables training, then a member saves the rule.
|
|
3158
2810
|
*/
|
|
3159
2811
|
| 'training_not_enabled';
|
|
3160
|
-
/**
|
|
3161
|
-
* How a training rule trains, on the wire of the automatic-training routes.
|
|
3162
|
-
*
|
|
3163
|
-
* DELIBERATELY NOT {@link TrainingMethod}. That union -- 'sft' | 'pt' -- is
|
|
3164
|
-
* training-service's, and the two services enumerate different things: a
|
|
3165
|
-
* training job can be a plain pre-training run, and a training rule cannot,
|
|
3166
|
-
* while a rule may ask for preference training and a job asks for that through
|
|
3167
|
-
* a different field. The values here are the loop service's own constants
|
|
3168
|
-
* (TrainingMethodSFT / TrainingMethodRLHF in
|
|
3169
|
-
* services/loop-service/cmd/training_types.go), which is what these routes
|
|
3170
|
-
* accept and return. Sharing the training-service union here advertised 'pt',
|
|
3171
|
-
* which this service refuses, and made 'rlhf', which it returns, unspellable.
|
|
3172
|
-
*
|
|
3173
|
-
* `rlhf_type` stays a plain string on purpose: the platform refuses it until
|
|
3174
|
-
* capabilities enable preference training, and an SDK union would have to be
|
|
3175
|
-
* republished to keep up with a server-side capability flag.
|
|
3176
|
-
*/
|
|
3177
|
-
export type LoopTrainingMethod = 'sft' | 'rlhf';
|
|
3178
|
-
/**
|
|
3179
|
-
* What a rule's stored `train_type` may READ as.
|
|
3180
|
-
*
|
|
3181
|
-
* Still three, and deliberately: the column has always allowed `full`, so a
|
|
3182
|
-
* rule created before standing rules were narrowed to adapters can still come
|
|
3183
|
-
* back carrying it. Narrowing the response type would make this SDK
|
|
3184
|
-
* misrepresent a row that really exists.
|
|
3185
|
-
*/
|
|
3186
|
-
export type TrainType = 'lora' | 'qlora' | 'full';
|
|
3187
|
-
/**
|
|
3188
|
-
* What a rule may be SET to, which is narrower.
|
|
3189
|
-
*
|
|
3190
|
-
* loop-service refuses `full` on preflight, create and update for a standing
|
|
3191
|
-
* rule: a rule trains unattended and on repeat, and a full fine-tune rewrites
|
|
3192
|
-
* every weight instead of adding a small adapter, so it cannot be compared or
|
|
3193
|
-
* rolled back cheaply. Typing the request as the wider set handed callers a
|
|
3194
|
-
* value guaranteed to come back a 400. One-off full fine-tunes are unaffected
|
|
3195
|
-
* — they are a different API.
|
|
3196
|
-
*/
|
|
3197
|
-
export type RuleTrainType = 'lora' | 'qlora';
|
|
3198
|
-
export type GradersScope = 'source' | 'all' | 'none';
|
|
3199
2812
|
/**
|
|
3200
2813
|
* Why a run fired. `variant` is a recipe variant: the pipeline had nothing new
|
|
3201
2814
|
* to learn, so it trained the same base model on the same conversations with
|
|
@@ -3226,7 +2839,6 @@ export type EvaluationItemStatus = 'pending' | 'working' | 'done' | 'failed';
|
|
|
3226
2839
|
export type EvaluationWinner = 'candidate' | 'incumbent' | 'tie';
|
|
3227
2840
|
/** Typed warning codes. Each surface renders its own plain sentence. */
|
|
3228
2841
|
export type EvaluationWarningCode = 'holdout_too_small' | 'judge_unreliable' | 'graders_not_applied' | 'many_failures' | 'no_judge' | 'judge_labelled_training_rows' | 'capture_source_moved';
|
|
3229
|
-
export type ConsentVia = 'console' | 'api_key' | 'sdk' | 'mcp';
|
|
3230
2842
|
/** A function an assistant turn asked to invoke. `arguments` is JSON-encoded. */
|
|
3231
2843
|
export interface TrainingToolCall {
|
|
3232
2844
|
id?: string;
|
|
@@ -3251,386 +2863,6 @@ export interface TrainingRuleServing {
|
|
|
3251
2863
|
/** Set only when kind is 'deployment'. */
|
|
3252
2864
|
deployment_id: string | null;
|
|
3253
2865
|
}
|
|
3254
|
-
/** One rung of a ranked GPU ladder. The provider is the neutral public brand. */
|
|
3255
|
-
export interface TrainingGPURung {
|
|
3256
|
-
gpu_type: string;
|
|
3257
|
-
gpu_count: number;
|
|
3258
|
-
provider: string;
|
|
3259
|
-
region: string;
|
|
3260
|
-
tier: string;
|
|
3261
|
-
}
|
|
3262
|
-
/** The consent object: one standing instruction to train, judge and possibly promote. */
|
|
3263
|
-
export interface TrainingRule {
|
|
3264
|
-
id: string;
|
|
3265
|
-
workspace_id: string;
|
|
3266
|
-
build_rule_id: string;
|
|
3267
|
-
name: string;
|
|
3268
|
-
enabled: boolean;
|
|
3269
|
-
/** Why the platform stopped firing this rule. Waiting must not look like broken. */
|
|
3270
|
-
paused_reason: TrainingRulePausedReason | null;
|
|
3271
|
-
cadence: TrainingCadence;
|
|
3272
|
-
cadence_hour_utc: number;
|
|
3273
|
-
cadence_weekday: number | null;
|
|
3274
|
-
combinator: TrainingCombinator;
|
|
3275
|
-
min_new_rows: number | null;
|
|
3276
|
-
/** Spreads firings across the hour so every daily rule does not land on one minute. */
|
|
3277
|
-
jitter_seconds: number;
|
|
3278
|
-
next_due_at: string | null;
|
|
3279
|
-
last_checked_at: string | null;
|
|
3280
|
-
last_fired_at: string | null;
|
|
3281
|
-
last_run_id: string | null;
|
|
3282
|
-
/** Rendered verbatim: "waiting: ...", "due, but ...", "fired: run 7 from ...". */
|
|
3283
|
-
last_reason: string | null;
|
|
3284
|
-
serving: TrainingRuleServing;
|
|
3285
|
-
model_id: string;
|
|
3286
|
-
/** A 40-hex commit, pinned when the rule is saved. */
|
|
3287
|
-
model_revision: string;
|
|
3288
|
-
training_method: LoopTrainingMethod;
|
|
3289
|
-
/** Refused by the platform until capabilities enable preference training. */
|
|
3290
|
-
rlhf_type: string | null;
|
|
3291
|
-
train_type: TrainType;
|
|
3292
|
-
config: Record<string, unknown>;
|
|
3293
|
-
train_gpu_priorities: TrainingGPURung[];
|
|
3294
|
-
train_max_price_hour_cents: number;
|
|
3295
|
-
deploy_gpu_priorities: TrainingGPURung[];
|
|
3296
|
-
deploy_max_price_hour_cents: number;
|
|
3297
|
-
training_ceiling_cents: number;
|
|
3298
|
-
candidate_ceiling_cents: number;
|
|
3299
|
-
/** A dollar figure, never a call count. */
|
|
3300
|
-
eval_ceiling_cents: number;
|
|
3301
|
-
monthly_ceiling_cents: number | null;
|
|
3302
|
-
eval_judge_id: string | null;
|
|
3303
|
-
eval_judge_model: string | null;
|
|
3304
|
-
eval_graders_scope: GradersScope;
|
|
3305
|
-
eval_max_rows: number;
|
|
3306
|
-
eval_max_tokens: number;
|
|
3307
|
-
min_holdout_rows: number;
|
|
3308
|
-
/**
|
|
3309
|
-
* How many versions this pipeline may make, 1 to 10 (five unless changed).
|
|
3310
|
-
* A version is a trained model whose comparison was reported having scored
|
|
3311
|
-
* at least one conversation, or one a member put live; a run that failed,
|
|
3312
|
-
* was cancelled or compared nothing is not one and does not count. At the
|
|
3313
|
-
* limit the pipeline stops, and a manual run is refused with
|
|
3314
|
-
* `409 VERSION_LIMIT_REACHED`, until the limit is raised.
|
|
3315
|
-
*/
|
|
3316
|
-
max_versions: number;
|
|
3317
|
-
/**
|
|
3318
|
-
* Try other training recipes when there is nothing new to learn.
|
|
3319
|
-
*
|
|
3320
|
-
* MONEY-BEARING. When on, a pipeline that has made at least one version, and
|
|
3321
|
-
* has nothing reviewed since it that adds anything new to train on (nothing
|
|
3322
|
-
* reviewed at all, or only reviews -- thumbs-down with no correction, say --
|
|
3323
|
-
* that give it no row the last version's set lacked), trains the same base
|
|
3324
|
-
* model on the same conversations with ONE setting changed, and that model
|
|
3325
|
-
* takes over only if it beats what serves the app, like any other version.
|
|
3326
|
-
* Each try is a full paid run inside the rule's per-run ceilings and its
|
|
3327
|
-
* monthly limit, so changing this in EITHER direction changes the terms: the
|
|
3328
|
-
* rule stops firing until the member accepts them again. False unless
|
|
3329
|
-
* somebody turned it on.
|
|
3330
|
-
*/
|
|
3331
|
-
explore_recipes: boolean;
|
|
3332
|
-
/**
|
|
3333
|
-
* The standing benchmark replayed on every run of this rule, beside the
|
|
3334
|
-
* per-run comparison and never instead of it. Null is the ordinary state.
|
|
3335
|
-
*
|
|
3336
|
-
* Set with `setTrainingRuleBenchmark`, not by an update to the rule: it is a
|
|
3337
|
-
* decision to replay a fixed set on every future run, and it has refusals of
|
|
3338
|
-
* its own. It raises no amount, so it does not invalidate consent.
|
|
3339
|
-
*/
|
|
3340
|
-
benchmark_id: string | null;
|
|
3341
|
-
/**
|
|
3342
|
-
* Whether that benchmark DECIDES the verdict, or only reports a number.
|
|
3343
|
-
*
|
|
3344
|
-
* False for every rule that has not asked, which is the point of it being
|
|
3345
|
-
* separate from `benchmark_id`. Attaching a benchmark means "replay this
|
|
3346
|
-
* fixed set on every run and put the score on the report"; it does not mean
|
|
3347
|
-
* "let that score overrule the comparison this rule was built on".
|
|
3348
|
-
*
|
|
3349
|
-
* When it is on it cuts both ways. A run whose held-out split came up short
|
|
3350
|
-
* can be decided at all -- before this, such a run trained a model, billed a
|
|
3351
|
-
* GPU and could never promote -- and a candidate that wins on fresh
|
|
3352
|
-
* conversations while losing ground on the fixed set is refused.
|
|
3353
|
-
*/
|
|
3354
|
-
benchmark_decides: boolean;
|
|
3355
|
-
/**
|
|
3356
|
-
* How many of the benchmark's pinned conversations must have scored before
|
|
3357
|
-
* it may decide anything. A replay that got through four of its forty has
|
|
3358
|
-
* not measured the new model.
|
|
3359
|
-
*/
|
|
3360
|
-
benchmark_min_rows: number;
|
|
3361
|
-
/**
|
|
3362
|
-
* The bar on the replay's own candidate-minus-incumbent delta. Like-for-like
|
|
3363
|
-
* WITHIN one replay -- both models, same pinned rows, same frozen judge, one
|
|
3364
|
-
* pass -- and not comparable between runs.
|
|
3365
|
-
*/
|
|
3366
|
-
benchmark_min_delta: number;
|
|
3367
|
-
auto_promote: boolean;
|
|
3368
|
-
promote_margin: number;
|
|
3369
|
-
promote_min_win_rate: number;
|
|
3370
|
-
/** A gate, not a badge. */
|
|
3371
|
-
min_judge_agreement: number;
|
|
3372
|
-
review_window_hours: number;
|
|
3373
|
-
candidate_boot_deadline_minutes: number;
|
|
3374
|
-
keep_candidate_warm_minutes: number;
|
|
3375
|
-
/** The member whose wallet pays, and the identity every peer call is stamped with. */
|
|
3376
|
-
authorized_by: string;
|
|
3377
|
-
terms_version: string;
|
|
3378
|
-
revision: number;
|
|
3379
|
-
/** Equal to revision only while the consent is current. */
|
|
3380
|
-
accepted_revision: number;
|
|
3381
|
-
accepted_at: string | null;
|
|
3382
|
-
deleted_at: string | null;
|
|
3383
|
-
created_at: string;
|
|
3384
|
-
updated_at: string;
|
|
3385
|
-
}
|
|
3386
|
-
/** One append-only record of a member agreeing to spend, with the words they read. */
|
|
3387
|
-
export interface TrainingRuleConsent {
|
|
3388
|
-
id: number;
|
|
3389
|
-
rule_id: string;
|
|
3390
|
-
workspace_id: string;
|
|
3391
|
-
revision: number;
|
|
3392
|
-
terms_version: string;
|
|
3393
|
-
terms_text: string;
|
|
3394
|
-
snapshot: Record<string, unknown>;
|
|
3395
|
-
accepted_by: string;
|
|
3396
|
-
accepted_via: ConsentVia;
|
|
3397
|
-
accepted_at: string;
|
|
3398
|
-
}
|
|
3399
|
-
export interface TrainingRulePreflightRefusal {
|
|
3400
|
-
stage: string;
|
|
3401
|
-
code: string;
|
|
3402
|
-
message: string;
|
|
3403
|
-
}
|
|
3404
|
-
/** One key that cannot follow a cutover, named so the caller can say which. */
|
|
3405
|
-
export interface TrainingRuleKeyRef {
|
|
3406
|
-
key_id: string;
|
|
3407
|
-
prefix: string;
|
|
3408
|
-
name: string;
|
|
3409
|
-
}
|
|
3410
|
-
export interface TrainingRuleKeyCheck {
|
|
3411
|
-
keys_missing_deployments_read: TrainingRuleKeyRef[];
|
|
3412
|
-
}
|
|
3413
|
-
/** The side-effect-free estimate, and the exact sentence the member will accept. */
|
|
3414
|
-
export interface TrainingRulePreflight {
|
|
3415
|
-
valid: boolean;
|
|
3416
|
-
model_revision: string;
|
|
3417
|
-
worst_hourly_training_cents: number;
|
|
3418
|
-
worst_hourly_candidate_cents: number;
|
|
3419
|
-
max_training_hours: number;
|
|
3420
|
-
max_candidate_hours: number;
|
|
3421
|
-
/** Informational. The fence on evaluation is the dollar ceiling, never this. */
|
|
3422
|
-
eval_calls_max: number;
|
|
3423
|
-
training_ceiling_cents: number;
|
|
3424
|
-
candidate_ceiling_cents: number;
|
|
3425
|
-
eval_ceiling_cents: number;
|
|
3426
|
-
warnings: string[];
|
|
3427
|
-
refusals: TrainingRulePreflightRefusal[];
|
|
3428
|
-
/**
|
|
3429
|
-
* Every peer the platform could not reach on this pass, one entry each, in
|
|
3430
|
-
* the same `{stage, code, message}` shape as a refusal.
|
|
3431
|
-
*
|
|
3432
|
-
* A warning, not a refusal: an unreachable peer judged nothing, so it never
|
|
3433
|
-
* refuses a save. It is still why the rule cannot fire — both
|
|
3434
|
-
* `worst_hourly_*` come back `0` — so `valid` is `false`. Key a "cannot
|
|
3435
|
-
* save" state on this when `refusals` is empty; the same sentences are also
|
|
3436
|
-
* in `warnings`, so render one or the other.
|
|
3437
|
-
*
|
|
3438
|
-
* Key it on THIS, not on `model_revision`. A degraded pass echoes back the
|
|
3439
|
-
* `model_revision` the request carried rather than emptying it, so
|
|
3440
|
-
* `if (!model_revision)` is false on exactly the passes it was meant to
|
|
3441
|
-
* catch. Use `unreachable.length > 0`, or `/^[0-9a-f]{40}$/`.
|
|
3442
|
-
*/
|
|
3443
|
-
unreachable: TrainingRulePreflightRefusal[];
|
|
3444
|
-
key_check: TrainingRuleKeyCheck;
|
|
3445
|
-
terms_text: string;
|
|
3446
|
-
terms_version: string;
|
|
3447
|
-
/**
|
|
3448
|
-
* The machines the hourly amounts are per hour of: the ladders the platform
|
|
3449
|
-
* chose when the request left them empty, or the caller's own echoed back
|
|
3450
|
-
* unchanged. An amount per hour means nothing without knowing what it is per
|
|
3451
|
-
* hour of, so show these beside the estimate before anyone accepts it.
|
|
3452
|
-
*/
|
|
3453
|
-
train_gpu_priorities: TrainingGPURung[];
|
|
3454
|
-
deploy_gpu_priorities: TrainingGPURung[];
|
|
3455
|
-
}
|
|
3456
|
-
export interface TrainingRuleBuildSpec {
|
|
3457
|
-
method: string;
|
|
3458
|
-
spec: Record<string, unknown>;
|
|
3459
|
-
}
|
|
3460
|
-
/**
|
|
3461
|
-
* The editable half of a build rule.
|
|
3462
|
-
*
|
|
3463
|
-
* Deliberately not `LoopBuildRuleParams`: that type requires `name` and
|
|
3464
|
-
* `method`, which the PUT route does not accept, so an edit written against it
|
|
3465
|
-
* would have to resend two fields the platform ignores and a caller could not
|
|
3466
|
-
* tell that changing them did nothing.
|
|
3467
|
-
*/
|
|
3468
|
-
export interface LoopBuildRuleUpdateParams {
|
|
3469
|
-
enabled?: boolean;
|
|
3470
|
-
/** Rows reviewed SINCE THE LAST BUILD before this fires again. Minimum 100. */
|
|
3471
|
-
min_new_rows?: number;
|
|
3472
|
-
/** The same selection `createDataset` takes, replayed verbatim. */
|
|
3473
|
-
spec?: LoopDatasetCreateParams;
|
|
3474
|
-
}
|
|
3475
|
-
/**
|
|
3476
|
-
* `?` means absent: leave that part of the sentence alone. `null` is only
|
|
3477
|
-
* allowed where the column is nullable and clearing it is a real edit, and it
|
|
3478
|
-
* is spelled out field by field rather than applied to the whole object.
|
|
3479
|
-
*/
|
|
3480
|
-
export interface TrainingRuleTriggerInput {
|
|
3481
|
-
cadence?: TrainingCadence;
|
|
3482
|
-
cadence_hour_utc?: number;
|
|
3483
|
-
/** null drops the weekday, which is what a weekly rule moved to daily needs. */
|
|
3484
|
-
cadence_weekday?: number | null;
|
|
3485
|
-
combinator?: TrainingCombinator;
|
|
3486
|
-
/**
|
|
3487
|
-
* Conversations reviewed since the last version before the rule fires on
|
|
3488
|
-
* rows. At least 100. null removes the row floor, leaving the schedule as the
|
|
3489
|
-
* only trigger.
|
|
3490
|
-
*/
|
|
3491
|
-
min_new_rows?: number | null;
|
|
3492
|
-
/** 1 to 10. Absent leaves it as it is; absent on a create means five. */
|
|
3493
|
-
max_versions?: number;
|
|
3494
|
-
/**
|
|
3495
|
-
* See {@link TrainingRule.explore_recipes}: every variant it allows is a
|
|
3496
|
-
* full paid run, so changing it -- on OR off -- is a money-bearing edit that
|
|
3497
|
-
* pauses the rule, and it makes no version of any kind, until the new terms
|
|
3498
|
-
* are accepted. Turning it off to save money stops the pipeline too, until
|
|
3499
|
-
* somebody accepts again. It sits beside `max_versions`
|
|
3500
|
-
* because both say when the pipeline makes another version. Absent leaves
|
|
3501
|
-
* it as it is; absent on a create means off.
|
|
3502
|
-
*/
|
|
3503
|
-
explore_recipes?: boolean;
|
|
3504
|
-
}
|
|
3505
|
-
export interface TrainingRuleTrainingInput {
|
|
3506
|
-
model_id?: string;
|
|
3507
|
-
model_revision?: string;
|
|
3508
|
-
training_method?: LoopTrainingMethod;
|
|
3509
|
-
/**
|
|
3510
|
-
* A plain string, refused by the platform until capabilities enable
|
|
3511
|
-
* preference training. null clears it, which is what moving a rule back to
|
|
3512
|
-
* plain supervised training means.
|
|
3513
|
-
*/
|
|
3514
|
-
rlhf_type?: string | null;
|
|
3515
|
-
train_type?: RuleTrainType;
|
|
3516
|
-
config?: Record<string, unknown>;
|
|
3517
|
-
train_gpu_priorities?: TrainingGPURung[];
|
|
3518
|
-
train_max_price_hour_cents?: number;
|
|
3519
|
-
}
|
|
3520
|
-
export interface TrainingRuleDeployInput {
|
|
3521
|
-
deploy_gpu_priorities?: TrainingGPURung[];
|
|
3522
|
-
deploy_max_price_hour_cents?: number;
|
|
3523
|
-
context_length?: number;
|
|
3524
|
-
quant?: string;
|
|
3525
|
-
serving_config?: Record<string, unknown>;
|
|
3526
|
-
hf_integration_id?: string;
|
|
3527
|
-
}
|
|
3528
|
-
export interface TrainingRuleMoneyInput {
|
|
3529
|
-
training_ceiling_cents?: number;
|
|
3530
|
-
candidate_ceiling_cents?: number;
|
|
3531
|
-
eval_ceiling_cents?: number;
|
|
3532
|
-
/** null removes the monthly cap. Absent leaves it exactly where it stands. */
|
|
3533
|
-
monthly_ceiling_cents?: number | null;
|
|
3534
|
-
/**
|
|
3535
|
-
* The most held-back conversations one comparison scores, 10 to 500 and
|
|
3536
|
-
* never below `min_holdout_rows`. Absent on a create means 100.
|
|
3537
|
-
*/
|
|
3538
|
-
eval_max_rows?: number;
|
|
3539
|
-
}
|
|
3540
|
-
export interface TrainingRuleEvaluationInput {
|
|
3541
|
-
/** null removes the judge from this rule. */
|
|
3542
|
-
judge_id?: string | null;
|
|
3543
|
-
/** null drops the override, returning to the judge's own advisory model. */
|
|
3544
|
-
judge_model?: string | null;
|
|
3545
|
-
graders_scope?: GradersScope;
|
|
3546
|
-
/**
|
|
3547
|
-
* The fewest held-back conversations that may decide a comparison. At least
|
|
3548
|
-
* 20, and 20 when absent on a create. A set that would hold back fewer is
|
|
3549
|
-
* not built.
|
|
3550
|
-
*/
|
|
3551
|
-
min_holdout_rows?: number;
|
|
3552
|
-
eval_max_tokens?: number;
|
|
3553
|
-
}
|
|
3554
|
-
export interface TrainingRulePromotionInput {
|
|
3555
|
-
auto_promote?: boolean;
|
|
3556
|
-
promote_margin?: number;
|
|
3557
|
-
promote_min_win_rate?: number;
|
|
3558
|
-
min_judge_agreement?: number;
|
|
3559
|
-
review_window_hours?: number;
|
|
3560
|
-
keep_candidate_warm_minutes?: number;
|
|
3561
|
-
}
|
|
3562
|
-
/**
|
|
3563
|
-
* The version of the terms the member read and accepted.
|
|
3564
|
-
*
|
|
3565
|
-
* Send back the `terms_version` the preflight returned. The SDK deliberately
|
|
3566
|
-
* ships no constant for it: a pinned version in a published package goes stale
|
|
3567
|
-
* the moment the platform revises the terms, and the value a member consented
|
|
3568
|
-
* to has to be the one they were actually shown.
|
|
3569
|
-
*/
|
|
3570
|
-
export interface TrainingRuleAcceptTerms {
|
|
3571
|
-
terms_version: string;
|
|
3572
|
-
}
|
|
3573
|
-
/** The create body without the yes. The preflight route takes exactly this. */
|
|
3574
|
-
export interface TrainingRulePreflightRequest {
|
|
3575
|
-
workspace_id?: string;
|
|
3576
|
-
name?: string;
|
|
3577
|
-
/** Exactly one of build_rule_id and build is given. */
|
|
3578
|
-
build_rule_id?: string;
|
|
3579
|
-
build?: TrainingRuleBuildSpec;
|
|
3580
|
-
trigger?: TrainingRuleTriggerInput;
|
|
3581
|
-
serving?: TrainingRuleServing;
|
|
3582
|
-
training?: TrainingRuleTrainingInput;
|
|
3583
|
-
deploy?: TrainingRuleDeployInput;
|
|
3584
|
-
money?: TrainingRuleMoneyInput;
|
|
3585
|
-
evaluation?: TrainingRuleEvaluationInput;
|
|
3586
|
-
promotion?: TrainingRulePromotionInput;
|
|
3587
|
-
enabled?: boolean;
|
|
3588
|
-
}
|
|
3589
|
-
export interface TrainingRuleCreateRequest extends TrainingRulePreflightRequest {
|
|
3590
|
-
/** Absent is a refusal, not a default. */
|
|
3591
|
-
accept_terms?: TrainingRuleAcceptTerms;
|
|
3592
|
-
/**
|
|
3593
|
-
* The standing benchmark this rule replays, set here so it applies to the
|
|
3594
|
-
* rule's FIRST run.
|
|
3595
|
-
*
|
|
3596
|
-
* A rule can fire within seconds of being created, and every step of a run
|
|
3597
|
-
* reads the benchmark off the rule as it stood at that moment. Attaching one
|
|
3598
|
-
* afterwards with `setTrainingRuleBenchmark` applies from the next run and
|
|
3599
|
-
* says nothing about the first, which is the run somebody is watching.
|
|
3600
|
-
*
|
|
3601
|
-
* Refused on the same terms the attach route refuses it: `404` when it is not
|
|
3602
|
-
* this workspace's, `409 BENCHMARK_RETIRED` when it has been retired. Use
|
|
3603
|
-
* `setTrainingRuleBenchmark` to change or remove it later; it is not on the
|
|
3604
|
-
* preflight body and not on the update body.
|
|
3605
|
-
*/
|
|
3606
|
-
benchmark_id?: string;
|
|
3607
|
-
/**
|
|
3608
|
-
* And whether that benchmark decides, for the same reason `benchmark_id` is
|
|
3609
|
-
* here: run 1 reads its gate off the snapshot it freezes, and a gate applied
|
|
3610
|
-
* by a second call lands after run 1 has been decided without it.
|
|
3611
|
-
*
|
|
3612
|
-
* `400 BENCHMARK_GATE_HAS_NO_SET` when it is true with no benchmark,
|
|
3613
|
-
* `400 BENCHMARK_GATE_NEEDS_MIN_ROWS` when it is true with no floor.
|
|
3614
|
-
*/
|
|
3615
|
-
benchmark_decides?: boolean;
|
|
3616
|
-
/** Cannot exceed the set's size: `400 BENCHMARK_GATE_UNREACHABLE`. */
|
|
3617
|
-
benchmark_min_rows?: number;
|
|
3618
|
-
/** Zero -- the default -- means "must not lose ground". */
|
|
3619
|
-
benchmark_min_delta?: number;
|
|
3620
|
-
}
|
|
3621
|
-
export interface TrainingRuleUpdateRequest extends TrainingRulePreflightRequest {
|
|
3622
|
-
expected_revision?: number;
|
|
3623
|
-
accept_terms?: TrainingRuleAcceptTerms;
|
|
3624
|
-
}
|
|
3625
|
-
export interface TrainingRuleConsentRequest {
|
|
3626
|
-
terms_version: string;
|
|
3627
|
-
revision: number;
|
|
3628
|
-
}
|
|
3629
|
-
export interface TrainingRuleListParams {
|
|
3630
|
-
enabled?: boolean;
|
|
3631
|
-
limit?: number;
|
|
3632
|
-
offset?: number;
|
|
3633
|
-
}
|
|
3634
2866
|
export interface TrainingRunPromoteRequest {
|
|
3635
2867
|
expected_revision?: number;
|
|
3636
2868
|
/** Required to promote an inconclusive comparison. */
|
|
@@ -3706,12 +2938,23 @@ export interface TrainingRun {
|
|
|
3706
2938
|
checkpoint_id: string | null;
|
|
3707
2939
|
checkpoint_step: number | null;
|
|
3708
2940
|
/**
|
|
3709
|
-
* Which
|
|
3710
|
-
*
|
|
3711
|
-
*
|
|
3712
|
-
|
|
2941
|
+
* Which attempt of its pipeline this run is: 1, 2, 3 ... as runs are made.
|
|
2942
|
+
* Null only for a run from before attempts were numbered that ended before
|
|
2943
|
+
* it competed.
|
|
2944
|
+
*/
|
|
2945
|
+
attempt_no: number | null;
|
|
2946
|
+
/**
|
|
2947
|
+
* Which version of the pipeline this run became. Set only when it was put
|
|
2948
|
+
* live -- promoted by the pipeline's rule or by a member -- as the previous
|
|
2949
|
+
* highest plus one; null for every attempt that did not become the live
|
|
2950
|
+
* model. A rollback renumbers nothing.
|
|
3713
2951
|
*/
|
|
3714
2952
|
version_no: number | null;
|
|
2953
|
+
/** The base model this run trained, and that model's family. */
|
|
2954
|
+
base_model_id: string | null;
|
|
2955
|
+
model_family: string | null;
|
|
2956
|
+
/** Machine time the run held (training job plus comparison machine); null until known. */
|
|
2957
|
+
gpu_seconds: number | null;
|
|
3715
2958
|
training_eval_loss: number | null;
|
|
3716
2959
|
candidate_deployment_id: string | null;
|
|
3717
2960
|
candidate_name: string | null;
|
|
@@ -3767,8 +3010,13 @@ export interface TrainingRunSummary {
|
|
|
3767
3010
|
error_code: string | null;
|
|
3768
3011
|
/** See TrainingRun.error_next_step. On the list row too: the Runs table is where a failure is first seen. */
|
|
3769
3012
|
error_next_step: string | null;
|
|
3013
|
+
/** See TrainingRun.attempt_no. */
|
|
3014
|
+
attempt_no: number | null;
|
|
3770
3015
|
/** See TrainingRun.version_no. */
|
|
3771
3016
|
version_no: number | null;
|
|
3017
|
+
/** See TrainingRun.base_model_id and model_family. */
|
|
3018
|
+
base_model_id: string | null;
|
|
3019
|
+
model_family: string | null;
|
|
3772
3020
|
verdict: TrainingVerdict | null;
|
|
3773
3021
|
decision: TrainingDecision | null;
|
|
3774
3022
|
train_rows: number | null;
|
|
@@ -3835,15 +3083,6 @@ export interface EvaluationDimensionScore {
|
|
|
3835
3083
|
candidate: number;
|
|
3836
3084
|
delta: number;
|
|
3837
3085
|
}
|
|
3838
|
-
export interface EvaluationGraderScore {
|
|
3839
|
-
grader_id: string;
|
|
3840
|
-
name: string;
|
|
3841
|
-
incumbent: number;
|
|
3842
|
-
candidate: number;
|
|
3843
|
-
delta: number;
|
|
3844
|
-
/** A grader that applied to four rows has not measured anything. */
|
|
3845
|
-
items_applied: number;
|
|
3846
|
-
}
|
|
3847
3086
|
export interface EvaluationWarning {
|
|
3848
3087
|
code: EvaluationWarningCode;
|
|
3849
3088
|
message: string;
|
|
@@ -3896,7 +3135,6 @@ export interface Evaluation {
|
|
|
3896
3135
|
judge_instructions: string | null;
|
|
3897
3136
|
judge_dimensions: EvaluationJudgeDimension[];
|
|
3898
3137
|
judge_system_prompt: string | null;
|
|
3899
|
-
grader_ids: string[];
|
|
3900
3138
|
decoding: EvaluationDecoding;
|
|
3901
3139
|
rows_selected: number;
|
|
3902
3140
|
rows_scored: number;
|
|
@@ -3911,11 +3149,7 @@ export interface Evaluation {
|
|
|
3911
3149
|
mean_delta: number | null;
|
|
3912
3150
|
judge_incumbent_mean: number | null;
|
|
3913
3151
|
judge_candidate_mean: number | null;
|
|
3914
|
-
grader_incumbent_mean: number | null;
|
|
3915
|
-
grader_candidate_mean: number | null;
|
|
3916
|
-
grader_items_applied: number;
|
|
3917
3152
|
per_dimension: EvaluationDimensionScore[];
|
|
3918
|
-
per_grader: EvaluationGraderScore[];
|
|
3919
3153
|
judge_agreement: JudgeAgreement | null;
|
|
3920
3154
|
/** Informational: there is no incumbent counterpart to compare it against. */
|
|
3921
3155
|
trainer_eval_loss: number | null;
|
|
@@ -3935,7 +3169,6 @@ export interface Evaluation {
|
|
|
3935
3169
|
export interface EvaluationItemSide {
|
|
3936
3170
|
completion: string | null;
|
|
3937
3171
|
tool_calls: TrainingToolCall[] | null;
|
|
3938
|
-
grader: Record<string, unknown> | null;
|
|
3939
3172
|
judge: Record<string, unknown> | null;
|
|
3940
3173
|
score: number | null;
|
|
3941
3174
|
model_version: string | null;
|
|
@@ -3960,34 +3193,18 @@ export interface EvaluationItem {
|
|
|
3960
3193
|
status: EvaluationItemStatus;
|
|
3961
3194
|
error: string | null;
|
|
3962
3195
|
}
|
|
3196
|
+
/**
|
|
3197
|
+
* Which pairs to read: the ones the attempt won (`candidate`), the ones the
|
|
3198
|
+
* version serving today won (`serving`; stored on the item as `incumbent`),
|
|
3199
|
+
* or the ties.
|
|
3200
|
+
*/
|
|
3201
|
+
export type EvaluationItemWinnerFilter = 'candidate' | 'serving' | 'tie';
|
|
3963
3202
|
export interface EvaluationItemListParams {
|
|
3964
|
-
winner?:
|
|
3965
|
-
/**
|
|
3203
|
+
winner?: EvaluationItemWinnerFilter;
|
|
3204
|
+
/** 50 when absent; at most 200. */
|
|
3966
3205
|
limit?: number;
|
|
3967
3206
|
offset?: number;
|
|
3968
3207
|
}
|
|
3969
|
-
export interface JudgeAgreementParams {
|
|
3970
|
-
/** RFC 3339. Narrows the window the agreement is computed over. */
|
|
3971
|
-
from?: string;
|
|
3972
|
-
to?: string;
|
|
3973
|
-
}
|
|
3974
|
-
/** Which model, whose words, how much. The cap is pushed to the workspace key. */
|
|
3975
|
-
export interface AgentSettings {
|
|
3976
|
-
workspace_id: string;
|
|
3977
|
-
default_model: string | null;
|
|
3978
|
-
judge_system_prompt: string | null;
|
|
3979
|
-
sampler_system_prompt: string | null;
|
|
3980
|
-
eval_monthly_cap_cents: number | null;
|
|
3981
|
-
updated_by: string | null;
|
|
3982
|
-
updated_at: string;
|
|
3983
|
-
}
|
|
3984
|
-
/** Absent leaves a setting alone; a present null returns it to the platform default. */
|
|
3985
|
-
export interface AgentSettingsRequest {
|
|
3986
|
-
default_model?: string | null;
|
|
3987
|
-
judge_system_prompt?: string | null;
|
|
3988
|
-
sampler_system_prompt?: string | null;
|
|
3989
|
-
eval_monthly_cap_cents?: number | null;
|
|
3990
|
-
}
|
|
3991
3208
|
/** A re-pointable public handle. Promotion is one row write here. */
|
|
3992
3209
|
export interface InferenceAlias {
|
|
3993
3210
|
workspace_id: string;
|
|
@@ -4012,8 +3229,11 @@ export interface InferenceAliasRequest {
|
|
|
4012
3229
|
* after it, and a delete throws the series away.
|
|
4013
3230
|
*/
|
|
4014
3231
|
export type BenchmarkStatus = 'active' | 'retired';
|
|
4015
|
-
/**
|
|
4016
|
-
|
|
3232
|
+
/**
|
|
3233
|
+
* Where the pinned conversations are copied from. Read once, at creation.
|
|
3234
|
+
* Always `traces`: the conversations named in `trace_ids`.
|
|
3235
|
+
*/
|
|
3236
|
+
export type BenchmarkSourceKind = 'traces';
|
|
4017
3237
|
/**
|
|
4018
3238
|
* The life of one replay.
|
|
4019
3239
|
*
|
|
@@ -4139,7 +3359,12 @@ export interface BenchmarkRun {
|
|
|
4139
3359
|
export interface BenchmarkHistoryPoint {
|
|
4140
3360
|
benchmark_run_id: string;
|
|
4141
3361
|
run_id: string;
|
|
3362
|
+
/** Counts every run in the workspace; label points with attempt_no and version. */
|
|
4142
3363
|
run_seq: number;
|
|
3364
|
+
/** The attempt of its pipeline this point measured. */
|
|
3365
|
+
attempt_no: number | null;
|
|
3366
|
+
/** The version that attempt became; null when it did not become one. */
|
|
3367
|
+
version: number | null;
|
|
4143
3368
|
rule_id: string | null;
|
|
4144
3369
|
rule_name: string | null;
|
|
4145
3370
|
created_at: string;
|
|
@@ -4151,12 +3376,15 @@ export interface BenchmarkHistoryPoint {
|
|
|
4151
3376
|
status: BenchmarkRunStatus;
|
|
4152
3377
|
status_reason: string | null;
|
|
4153
3378
|
}
|
|
4154
|
-
/**
|
|
3379
|
+
/**
|
|
3380
|
+
* Where the conversations are copied FROM. Read once, at creation, never again.
|
|
3381
|
+
* A benchmark pins conversations by id; to make one from a file, import it
|
|
3382
|
+
* (`importRows`, with no tag, so it never trains a pipeline) and pin the
|
|
3383
|
+
* `trace_ids` the import returns.
|
|
3384
|
+
*/
|
|
4155
3385
|
export interface BenchmarkSource {
|
|
4156
3386
|
kind: BenchmarkSourceKind;
|
|
4157
|
-
trace_ids
|
|
4158
|
-
dataset_id?: string;
|
|
4159
|
-
split?: string;
|
|
3387
|
+
trace_ids: string[];
|
|
4160
3388
|
}
|
|
4161
3389
|
/**
|
|
4162
3390
|
* The pin: 20 to 200 conversations, and a judge with a rubric to measure them.
|
|
@@ -4165,15 +3393,16 @@ export interface BenchmarkSource {
|
|
|
4165
3393
|
* Pinning 47 of the 50 somebody chose is the set being wrong from the first
|
|
4166
3394
|
* day, and they would never find out.
|
|
4167
3395
|
*/
|
|
3396
|
+
/**
|
|
3397
|
+
* A name and the conversations, nothing else. The platform freezes the judge
|
|
3398
|
+
* and the model it scores on, gives both models the comparison's own answer
|
|
3399
|
+
* length, and sets what one replay may spend; none of that is sent (the
|
|
3400
|
+
* service refuses `judge_id`, `judge_model`, `max_tokens` and
|
|
3401
|
+
* `per_run_ceiling_cents` by name).
|
|
3402
|
+
*/
|
|
4168
3403
|
export interface BenchmarkCreateParams {
|
|
4169
3404
|
name: string;
|
|
4170
|
-
judge_id: string;
|
|
4171
3405
|
source: BenchmarkSource;
|
|
4172
|
-
/** Absent uses the judge's own model, and absent that the workspace default. */
|
|
4173
|
-
judge_model?: string;
|
|
4174
|
-
/** What both models are given to answer in. A shorter answer is a different answer. */
|
|
4175
|
-
max_tokens?: number;
|
|
4176
|
-
per_run_ceiling_cents: number;
|
|
4177
3406
|
}
|
|
4178
3407
|
export interface BenchmarkListParams {
|
|
4179
3408
|
status?: BenchmarkStatus;
|
|
@@ -4181,11 +3410,12 @@ export interface BenchmarkListParams {
|
|
|
4181
3410
|
offset?: number;
|
|
4182
3411
|
}
|
|
4183
3412
|
export interface BenchmarkItemListParams {
|
|
4184
|
-
/**
|
|
3413
|
+
/** 50 when absent; at most 200. */
|
|
4185
3414
|
limit?: number;
|
|
4186
3415
|
offset?: number;
|
|
4187
3416
|
}
|
|
4188
3417
|
export interface BenchmarkHistoryParams {
|
|
3418
|
+
/** 50 when absent; at most 200. */
|
|
4189
3419
|
limit?: number;
|
|
4190
3420
|
offset?: number;
|
|
4191
3421
|
}
|
|
@@ -4194,77 +3424,96 @@ export interface BenchmarkRetireParams {
|
|
|
4194
3424
|
reason?: string;
|
|
4195
3425
|
}
|
|
4196
3426
|
/**
|
|
4197
|
-
*
|
|
4198
|
-
*
|
|
4199
|
-
*
|
|
4200
|
-
*
|
|
4201
|
-
* field here. This is the route that attaches a benchmark AND the route that
|
|
4202
|
-
* moves its bar a month later, so a call naming only `benchmark_id` must not
|
|
4203
|
-
* reset a floor somebody chose, and a call naming only `benchmark_min_delta`
|
|
4204
|
-
* must not detach the set.
|
|
4205
|
-
*
|
|
4206
|
-
* Detaching is the one exception: a present null `benchmark_id` clears
|
|
4207
|
-
* `benchmark_decides` with it, because a gate with nothing behind it is a
|
|
4208
|
-
* verdict waiting on a measurement that will never come. Asking for both in
|
|
4209
|
-
* one request -- null id and `benchmark_decides: true` -- is refused rather
|
|
4210
|
-
* than resolved, because either resolution is a guess about which half you
|
|
4211
|
-
* meant.
|
|
4212
|
-
*/
|
|
4213
|
-
export interface TrainingRuleBenchmarkRequest {
|
|
4214
|
-
benchmark_id?: string | null;
|
|
4215
|
-
benchmark_decides?: boolean;
|
|
4216
|
-
benchmark_min_rows?: number;
|
|
4217
|
-
benchmark_min_delta?: number;
|
|
4218
|
-
}
|
|
4219
|
-
/**
|
|
4220
|
-
* The machine codes `runTrainingRule` can be refused with. Each is the
|
|
4221
|
-
* platform declining to start a paid run it would only refuse a minute later,
|
|
4222
|
-
* so each is said in front of the caller rather than left for the rule's
|
|
4223
|
-
* `last_reason` to explain after they have stopped looking.
|
|
3427
|
+
* The machine codes `runPipeline` (train now) can be refused with: every one
|
|
3428
|
+
* the run route answers (the SDK's tests read them off the service). Each is
|
|
3429
|
+
* the platform declining to start a paid attempt it would only refuse a minute
|
|
3430
|
+
* later, so each is said in front of the caller.
|
|
4224
3431
|
*
|
|
4225
|
-
* - `
|
|
4226
|
-
*
|
|
4227
|
-
*
|
|
3432
|
+
* - `NOT_ENOUGH_SAMPLES` (422): fewer samples than the next attempt needs.
|
|
3433
|
+
* The body carries `have` and `need`. Train now skips the schedule, never
|
|
3434
|
+
* the minimum.
|
|
3435
|
+
* - `PIPELINE_NEEDS_SAVE` (409): the pipeline's settings changed without a
|
|
3436
|
+
* pipeline save since; save them (saving is the authorization) and train
|
|
3437
|
+
* again.
|
|
3438
|
+
* - `RULE_PAUSED` (409): the pipeline is switched off or the platform paused
|
|
3439
|
+
* it (`paused_reason` says why).
|
|
3440
|
+
* - `RUN_ACTIVE` (409): one attempt at a time, so two cannot race to promote.
|
|
4228
3441
|
* - `VERSION_LIMIT_REACHED` (409): the pipeline has made `max_versions`
|
|
4229
|
-
* versions. Raise the limit or
|
|
4230
|
-
* - `MONTHLY_LIMIT_REACHED` (409): the most this month's
|
|
4231
|
-
* plus the most one more
|
|
4232
|
-
* The body carries the figures:
|
|
3442
|
+
* versions. Raise the limit or remove it.
|
|
3443
|
+
* - `MONTHLY_LIMIT_REACHED` (409): the most this month's attempts can have
|
|
3444
|
+
* cost plus the most one more may cost would pass the pipeline's internal
|
|
3445
|
+
* monthly limit. The body carries the figures:
|
|
4233
3446
|
* see {@link TrainingRuleMonthlyLimitRefusal}.
|
|
4234
|
-
* - `
|
|
4235
|
-
*
|
|
4236
|
-
*
|
|
4237
|
-
*
|
|
4238
|
-
* - `
|
|
4239
|
-
*
|
|
4240
|
-
|
|
4241
|
-
|
|
4242
|
-
|
|
4243
|
-
*
|
|
4244
|
-
*
|
|
4245
|
-
* that switches a rule on or into preference training or carries
|
|
4246
|
-
* `accept_terms`, and `consentTrainingRule`. Every other preflight refusal is
|
|
4247
|
-
* a `400` carrying the peer's own sentence; `preflightTrainingRule` answers
|
|
4248
|
-
* `200` with the same codes in `refusals`.
|
|
3447
|
+
* - `SCORING_CAPPED` (409): the key the comparison runs under is at its
|
|
3448
|
+
* monthly cap, so what this attempt trained could not be scored. The body
|
|
3449
|
+
* carries `resumes_at`, when the cap lifts (the next UTC month, or sooner
|
|
3450
|
+
* if the key's cap is raised).
|
|
3451
|
+
* - `AGENT_COULD_NOT_START` (503): the key every comparison is scored with is
|
|
3452
|
+
* missing or was refused, and train now could not renew it just then.
|
|
3453
|
+
* Nothing was started. Press train now again in a minute.
|
|
3454
|
+
* - `AGENT_OFF` (409): scoring could not run: that key was still unusable
|
|
3455
|
+
* after train now renewed it, so no attempt was started. Press train now
|
|
3456
|
+
* again in a minute. Nothing in the pipeline's settings causes it, so
|
|
3457
|
+
* saving the pipeline does not clear it.
|
|
4249
3458
|
*
|
|
4250
|
-
*
|
|
4251
|
-
*
|
|
4252
|
-
*
|
|
4253
|
-
* the organisation. It is answered ahead of every other refusal. A rule that
|
|
4254
|
-
* meets it at fire time is paused `training_not_enabled` instead (see
|
|
4255
|
-
* {@link TrainingRulePausedReason}).
|
|
4256
|
-
* - `METHOD_NOT_ENABLED` (422): a training method the platform has not enabled.
|
|
4257
|
-
* - `CEILING_TOO_LOW` (422): a ceiling cannot buy the minimum hours at the
|
|
4258
|
-
* accepted price cap.
|
|
3459
|
+
* Only the last two clear on their own within a minute, and
|
|
3460
|
+
* `MONTHLY_LIMIT_REACHED` and `SCORING_CAPPED` once their date has passed;
|
|
3461
|
+
* retrying any other one unchanged gets the same answer.
|
|
4259
3462
|
*/
|
|
4260
|
-
export type
|
|
3463
|
+
export type TrainingRuleRunRefusalCode = 'NOT_ENOUGH_SAMPLES' | 'PIPELINE_NEEDS_SAVE' | 'RULE_PAUSED' | 'RUN_ACTIVE' | 'VERSION_LIMIT_REACHED' | 'MONTHLY_LIMIT_REACHED' | 'SCORING_CAPPED' | 'AGENT_COULD_NOT_START' | 'AGENT_OFF';
|
|
4261
3464
|
/**
|
|
4262
|
-
* The machine codes `
|
|
3465
|
+
* The machine codes saving a pipeline (`createPipeline` or `updatePipeline`)
|
|
3466
|
+
* can be refused with: every one the two routes answer beyond a malformed
|
|
3467
|
+
* body (400, which names the field). The SDK's tests read them off the
|
|
3468
|
+
* service. Nothing is saved when any of them is answered.
|
|
4263
3469
|
*
|
|
4264
|
-
*
|
|
4265
|
-
*
|
|
3470
|
+
* The name and tags:
|
|
3471
|
+
* - `PIPELINE_NAME_TAKEN` (409): another pipeline in the workspace has the
|
|
3472
|
+
* name, a deleted one included. Choose another name.
|
|
3473
|
+
* - `OWN_TAG_TAKEN` (409): another pipeline already owns the tag this name
|
|
3474
|
+
* makes. Choose another name.
|
|
3475
|
+
*
|
|
3476
|
+
* An edit only:
|
|
3477
|
+
* - `REVISION_MISMATCH` (409): `expected_revision` is not the pipeline's
|
|
3478
|
+
* revision any more. Read it again and resend.
|
|
3479
|
+
* - `RUN_ACTIVE` (409): the model cannot change while an attempt is running.
|
|
3480
|
+
* - `NOT_A_SIMPLE_PIPELINE` (409): a tag, use case or challenger edit on a
|
|
3481
|
+
* pipeline created before pipelines had them; its other settings can
|
|
3482
|
+
* still change.
|
|
3483
|
+
*
|
|
3484
|
+
* The model and when to train:
|
|
3485
|
+
* - `MODEL_NOT_TRAINABLE` (422): `model.id` or `challenger_model.id` is not a
|
|
3486
|
+
* model `listModels` offers, or the challenger is the model itself.
|
|
3487
|
+
* - `MIN_SAMPLES_TOO_LOW` (422): `train_when.min_samples` is under the floor
|
|
3488
|
+
* `listModels` returns as `floors.new_samples`.
|
|
3489
|
+
* - `NO_GPU_FITS` (422): no machines can train or serve the model right now.
|
|
3490
|
+
* - `MODEL_REVISION_UNAVAILABLE` (422): the model could not be pinned to an
|
|
3491
|
+
* exact revision because the training service did not answer. Try again
|
|
3492
|
+
* once it does.
|
|
3493
|
+
* - `NO_JUDGE_MODEL` (422): no model the workspace can call is able to score
|
|
3494
|
+
* the attempts, so none could ever be measured.
|
|
3495
|
+
* - `PROMOTION_POLICY_UNKNOWN`, `PROMOTION_K_REQUIRED`,
|
|
3496
|
+
* `PROMOTION_K_UNREACHABLE`, `PROMOTION_POLICY_NEEDS_BENCHMARK` (422): the
|
|
3497
|
+
* promotion policy is not one of the six, `k_of_n` came without `k`, `k`
|
|
3498
|
+
* is more than the measurements the pipeline takes, or the policy needs a
|
|
3499
|
+
* benchmark the pipeline does not have yet.
|
|
3500
|
+
*
|
|
3501
|
+
* The platform:
|
|
3502
|
+
* - `TRAINING_NOT_ENABLED` (409): training is in beta and this organisation
|
|
3503
|
+
* does not hold the grant. Nothing in the request can lift it, so do not
|
|
3504
|
+
* retry: ask the platform team to enable training for the organisation. A
|
|
3505
|
+
* pipeline that meets it at fire time is paused `training_not_enabled`
|
|
3506
|
+
* instead (see {@link TrainingRulePausedReason}).
|
|
3507
|
+
* - `METHOD_NOT_ENABLED` (422): the platform has switched off the training
|
|
3508
|
+
* method pipelines train with.
|
|
3509
|
+
* - `CEILING_TOO_LOW` (422): the platform's own runaway limits cannot cover
|
|
3510
|
+
* the machines it chose for the model. Nothing in the request sets them.
|
|
3511
|
+
* - `MODELS_UNAVAILABLE`, `GPU_OPTIONS_UNAVAILABLE`, `AGENT_COULD_NOT_START`
|
|
3512
|
+
* (503): a service the save needs (the model list, the machine list, or
|
|
3513
|
+
* the key every comparison is scored with) did not answer just then. Try
|
|
3514
|
+
* again in a minute.
|
|
4266
3515
|
*/
|
|
4267
|
-
export type
|
|
3516
|
+
export type TrainingRuleSaveRefusalCode = 'PIPELINE_NAME_TAKEN' | 'OWN_TAG_TAKEN' | 'REVISION_MISMATCH' | 'RUN_ACTIVE' | 'NOT_A_SIMPLE_PIPELINE' | 'MODEL_NOT_TRAINABLE' | 'MIN_SAMPLES_TOO_LOW' | 'NO_GPU_FITS' | 'MODEL_REVISION_UNAVAILABLE' | 'NO_JUDGE_MODEL' | 'PROMOTION_POLICY_UNKNOWN' | 'PROMOTION_K_REQUIRED' | 'PROMOTION_K_UNREACHABLE' | 'PROMOTION_POLICY_NEEDS_BENCHMARK' | 'TRAINING_NOT_ENABLED' | 'METHOD_NOT_ENABLED' | 'CEILING_TOO_LOW' | 'MODELS_UNAVAILABLE' | 'GPU_OPTIONS_UNAVAILABLE' | 'AGENT_COULD_NOT_START';
|
|
4268
3517
|
/**
|
|
4269
3518
|
* The body of a manual run refused `409 MONTHLY_LIMIT_REACHED`, as
|
|
4270
3519
|
* `ApiError.body` carries it.
|
|
@@ -4293,7 +3542,7 @@ export interface TrainingRuleMonthlyLimitRefusal {
|
|
|
4293
3542
|
resumes_at: string;
|
|
4294
3543
|
}
|
|
4295
3544
|
/** The most versions a pipeline may be set to make (loop-service maxVersionsCeiling). */
|
|
4296
|
-
export declare const PIPELINE_MAX_VERSIONS_CEILING =
|
|
3545
|
+
export declare const PIPELINE_MAX_VERSIONS_CEILING = 100;
|
|
4297
3546
|
/**
|
|
4298
3547
|
* What a pipeline is doing. `needs_review` and `needs_funds` are a finished
|
|
4299
3548
|
* version waiting on the MEMBER -- a decision, or a top-up -- which is not the
|
|
@@ -4305,18 +3554,29 @@ export declare const PIPELINE_MAX_VERSIONS_CEILING = 10;
|
|
|
4305
3554
|
*/
|
|
4306
3555
|
export type PipelineStatus = 'running' | 'needs_review' | 'needs_funds' | 'waiting' | 'paused' | 'complete';
|
|
4307
3556
|
/**
|
|
4308
|
-
* The
|
|
4309
|
-
*
|
|
3557
|
+
* The attempt in flight. It is not a version until it is put live -- by the
|
|
3558
|
+
* pipeline's promotion rule or by a member.
|
|
4310
3559
|
*/
|
|
4311
3560
|
export interface PipelineActiveRun {
|
|
4312
3561
|
run_id: string;
|
|
3562
|
+
/** Its attempt number within the pipeline. */
|
|
3563
|
+
attempt: number | null;
|
|
3564
|
+
/** Why it was made; see {@link PipelineAttempt.trigger}. */
|
|
3565
|
+
trigger: PipelineAttemptTrigger;
|
|
3566
|
+
/** The base model it trains, and that model's family. */
|
|
3567
|
+
base_model_id: string | null;
|
|
3568
|
+
model_family: string | null;
|
|
4313
3569
|
state: TrainingRunState;
|
|
4314
3570
|
since: string;
|
|
4315
|
-
/**
|
|
4316
|
-
|
|
4317
|
-
|
|
4318
|
-
|
|
3571
|
+
/** When the attempt was made. */
|
|
3572
|
+
started_at: string;
|
|
3573
|
+
/** Machine time it has held so far; null until known. */
|
|
3574
|
+
gpu_seconds: number | null;
|
|
3575
|
+
/** What it has been billed so far, in cents. */
|
|
3576
|
+
cost_cents: number;
|
|
3577
|
+
/** The version number it will take if it is put live. */
|
|
4319
3578
|
will_be_version: number;
|
|
3579
|
+
/** Always null: nothing in flight is a version yet. */
|
|
4320
3580
|
version_no: number | null;
|
|
4321
3581
|
last_reason: string | null;
|
|
4322
3582
|
last_error: string | null;
|
|
@@ -4332,20 +3592,29 @@ export interface PipelineActiveRun {
|
|
|
4332
3592
|
recipe_note: string | null;
|
|
4333
3593
|
}
|
|
4334
3594
|
/**
|
|
4335
|
-
* One version:
|
|
4336
|
-
*
|
|
4337
|
-
*
|
|
3595
|
+
* One version: an attempt that was put live -- promoted by the pipeline's
|
|
3596
|
+
* rule or by a member. Only winners are versions, numbered v1, v2, ... in the
|
|
3597
|
+
* order they went live; numbers only go up and a rollback renumbers nothing.
|
|
4338
3598
|
*/
|
|
4339
3599
|
export interface PipelineVersion {
|
|
4340
3600
|
version: number;
|
|
4341
3601
|
run_id: string;
|
|
3602
|
+
/** The attempt that became this version. */
|
|
3603
|
+
attempt: number | null;
|
|
3604
|
+
/** The base model it was trained from, and that model's family. */
|
|
3605
|
+
base_model_id: string | null;
|
|
3606
|
+
model_family: string | null;
|
|
4342
3607
|
seq: number;
|
|
4343
3608
|
state: TrainingRunState;
|
|
4344
3609
|
verdict: TrainingVerdict | null;
|
|
4345
3610
|
decision: TrainingDecision | null;
|
|
4346
3611
|
decided_at: string | null;
|
|
3612
|
+
/** True for the version the app is served by today (the same as `is_champion`). */
|
|
3613
|
+
is_live: boolean;
|
|
4347
3614
|
/** True for the version the app is served by today. */
|
|
4348
3615
|
is_champion: boolean;
|
|
3616
|
+
/** When it was first put live. */
|
|
3617
|
+
promoted_at: string | null;
|
|
4349
3618
|
checkpoint_id: string | null;
|
|
4350
3619
|
train_rows: number | null;
|
|
4351
3620
|
holdout_rows: number | null;
|
|
@@ -4379,6 +3648,8 @@ export interface PipelineVersion {
|
|
|
4379
3648
|
* not what it did.
|
|
4380
3649
|
*/
|
|
4381
3650
|
cost_includes_candidate_ceiling: boolean;
|
|
3651
|
+
/** Machine time the attempt held (training and comparison); null when unknown. */
|
|
3652
|
+
gpu_seconds: number | null;
|
|
4382
3653
|
created_at: string;
|
|
4383
3654
|
/**
|
|
4384
3655
|
* When this version's head-to-head started. The champion it faced is the
|
|
@@ -4403,42 +3674,129 @@ export interface PipelineVersion {
|
|
|
4403
3674
|
finished_at: string | null;
|
|
4404
3675
|
}
|
|
4405
3676
|
/**
|
|
4406
|
-
* A
|
|
4407
|
-
* which one serves, whether each beat the one before it, and what
|
|
4408
|
-
* next. READ-ONLY: every decision stays on the route that owns it
|
|
4409
|
-
* reject and roll back on the
|
|
3677
|
+
* A pipeline: one use case, and the series of versions it produces -- which
|
|
3678
|
+
* exist, which one serves, whether each beat the one before it, and what
|
|
3679
|
+
* happens next. READ-ONLY: every decision stays on the route that owns it
|
|
3680
|
+
* (promote, reject and roll back on the attempt; the version limit is set by
|
|
3681
|
+
* editing the pipeline).
|
|
4410
3682
|
*/
|
|
4411
3683
|
/**
|
|
4412
|
-
*
|
|
4413
|
-
*
|
|
4414
|
-
*
|
|
4415
|
-
|
|
4416
|
-
|
|
4417
|
-
|
|
4418
|
-
|
|
4419
|
-
|
|
4420
|
-
|
|
4421
|
-
|
|
4422
|
-
|
|
4423
|
-
|
|
4424
|
-
|
|
4425
|
-
|
|
4426
|
-
|
|
3684
|
+
* Why an attempt was made: new data (`data`, a recipe variant included), the
|
|
3685
|
+
* schedule (`schedule`), a member's train-now (`manual`), or the challenger
|
|
3686
|
+
* model trained on the same data as the attempt before it (`challenger`).
|
|
3687
|
+
*/
|
|
3688
|
+
export type PipelineAttemptTrigger = 'data' | 'schedule' | 'manual' | 'challenger';
|
|
3689
|
+
/** What became of an attempt. */
|
|
3690
|
+
export type PipelineAttemptOutcome = 'became_version' | 'not_better' | 'inconclusive' | 'stopped' | 'failed' | 'running' | 'waiting_for_review';
|
|
3691
|
+
/** One attempt: every training run a pipeline makes, winner or not. */
|
|
3692
|
+
export interface PipelineAttempt {
|
|
3693
|
+
attempt: number;
|
|
3694
|
+
run_id: string;
|
|
3695
|
+
state: TrainingRunState;
|
|
3696
|
+
trigger: PipelineAttemptTrigger;
|
|
3697
|
+
base_model_id: string | null;
|
|
3698
|
+
model_family: string | null;
|
|
3699
|
+
outcome: PipelineAttemptOutcome;
|
|
3700
|
+
/** The version it became, when it became one. */
|
|
3701
|
+
version: number | null;
|
|
3702
|
+
created_at: string;
|
|
3703
|
+
finished_at: string | null;
|
|
3704
|
+
gpu_seconds: number | null;
|
|
3705
|
+
/** As the versions are priced once it has ended; as recorded so far while it runs. */
|
|
3706
|
+
cost_cents: number;
|
|
3707
|
+
verdict: TrainingVerdict | null;
|
|
3708
|
+
win_rate: number | null;
|
|
3709
|
+
evaluation_id: string | null;
|
|
3710
|
+
/** Its last sentence. */
|
|
3711
|
+
reason: string | null;
|
|
3712
|
+
}
|
|
3713
|
+
/** A model and the exact commit of it. */
|
|
3714
|
+
export interface PipelineModelRef {
|
|
3715
|
+
id: string;
|
|
3716
|
+
revision: string;
|
|
3717
|
+
}
|
|
3718
|
+
/**
|
|
3719
|
+
* When a new version takes over. `holdout` (the default) when it wins on the
|
|
3720
|
+
* held-back set; `all`, `primary`, `k_of_n` (with `k`) and `weighted` also
|
|
3721
|
+
* weigh the pipeline's benchmarks; `manual` never on its own.
|
|
3722
|
+
*/
|
|
3723
|
+
export interface PipelinePromotion {
|
|
3724
|
+
policy: 'holdout' | 'all' | 'primary' | 'k_of_n' | 'weighted' | 'manual';
|
|
3725
|
+
k: number | null;
|
|
3726
|
+
}
|
|
3727
|
+
/** A pipeline's schedule, UTC: every day at an hour, or every week on a weekday at an hour. */
|
|
3728
|
+
export interface PipelineSchedule {
|
|
3729
|
+
every: 'day' | 'week';
|
|
3730
|
+
/** 0 (Sunday) to 6, for `week`; null for `day`. */
|
|
3731
|
+
weekday: number | null;
|
|
3732
|
+
hour_utc: number;
|
|
3733
|
+
}
|
|
3734
|
+
/**
|
|
3735
|
+
* When a pipeline trains: at least `min_samples` new samples AND, when there
|
|
3736
|
+
* is one, at its scheduled slot. A null schedule trains as soon as the minimum
|
|
3737
|
+
* is met.
|
|
3738
|
+
*/
|
|
3739
|
+
export interface PipelineTrainWhen {
|
|
3740
|
+
min_samples: number;
|
|
3741
|
+
schedule: PipelineSchedule | null;
|
|
3742
|
+
}
|
|
3743
|
+
/** What is held back beyond max(50, 5% of the data): a larger percent, or a count. */
|
|
3744
|
+
export interface PipelineHoldout {
|
|
3745
|
+
percent: number;
|
|
3746
|
+
count: number | null;
|
|
3747
|
+
min_rows: number;
|
|
3748
|
+
}
|
|
3749
|
+
/**
|
|
3750
|
+
* How far the data is from the next version. `first_version`: `needed` usable
|
|
3751
|
+
* rows v1 trains on. `new_rows`: `needed` usable rows the last version's set
|
|
3752
|
+
* did not have. `have`, `short` and `message` are measured on the detail only.
|
|
4427
3753
|
*/
|
|
4428
|
-
export
|
|
3754
|
+
export interface PipelineMinimums {
|
|
3755
|
+
basis: 'first_version' | 'new_rows';
|
|
3756
|
+
needed: number;
|
|
3757
|
+
have: number | null;
|
|
3758
|
+
short: number | null;
|
|
3759
|
+
message: string | null;
|
|
3760
|
+
}
|
|
4429
3761
|
export interface Pipeline {
|
|
4430
3762
|
rule_id: string;
|
|
4431
3763
|
rule_name: string;
|
|
4432
3764
|
enabled: boolean;
|
|
4433
|
-
/** The
|
|
4434
|
-
|
|
3765
|
+
/** The settings revision an edit sends back as `expected_revision`. */
|
|
3766
|
+
revision: number;
|
|
3767
|
+
/** The limit the member set, 1 to 100, or null for none. */
|
|
3768
|
+
max_versions: number | null;
|
|
4435
3769
|
/** Versions made so far. At `max_versions` the pipeline stops. */
|
|
4436
3770
|
versions_made: number;
|
|
3771
|
+
/** What the pipeline is for, in the member's words. Empty on a pipeline created before pipelines had a use case. */
|
|
3772
|
+
use_case: string;
|
|
3773
|
+
/** The tag imports into this pipeline are given. Null on a pipeline created before pipelines had a tag of their own. */
|
|
3774
|
+
own_tag: string | null;
|
|
3775
|
+
/** A conversation carrying ANY of these tags is this pipeline's data. */
|
|
3776
|
+
tags: string[];
|
|
3777
|
+
/**
|
|
3778
|
+
* A second base model trained on the same data after each attempt; it
|
|
3779
|
+
* becomes the next version only if it beats the live one. `revision` is the
|
|
3780
|
+
* commit the server pinned.
|
|
3781
|
+
*/
|
|
3782
|
+
challenger_model: PipelineModelRef | null;
|
|
3783
|
+
/** `sft` today; `kto`, `dpo` and `grpo` cannot be chosen yet. */
|
|
3784
|
+
training_type: string;
|
|
3785
|
+
/** How each version starts: `fresh`, a new adapter over the pinned base trained on all accepted data. */
|
|
3786
|
+
version_base: string;
|
|
3787
|
+
train_when: PipelineTrainWhen;
|
|
3788
|
+
promotion: PipelinePromotion;
|
|
3789
|
+
holdout: PipelineHoldout;
|
|
3790
|
+
/** Rows the next version holds back and never trains on. Detail only; null on the list. */
|
|
3791
|
+
holdout_size: number | null;
|
|
3792
|
+
minimums: PipelineMinimums;
|
|
4437
3793
|
/** The pinned base every version is a fresh adapter over, which is what makes their scores comparable. */
|
|
4438
3794
|
base_model_id: string;
|
|
4439
3795
|
base_model_revision: string;
|
|
4440
3796
|
train_type: string;
|
|
4441
|
-
/**
|
|
3797
|
+
/** The version serving now; null when what serves is not a version of this pipeline. */
|
|
3798
|
+
live_version: number | null;
|
|
3799
|
+
/** The same number under its older name. */
|
|
4442
3800
|
champion_version: number | null;
|
|
4443
3801
|
/** The name the app calls. It does not change when a version is promoted. */
|
|
4444
3802
|
serving_name: string;
|
|
@@ -4469,17 +3827,6 @@ export interface Pipeline {
|
|
|
4469
3827
|
* until somebody does.
|
|
4470
3828
|
*/
|
|
4471
3829
|
needs_consent: boolean;
|
|
4472
|
-
/** The rule's explore_recipes, as it stands. */
|
|
4473
|
-
explore_recipes: boolean;
|
|
4474
|
-
/**
|
|
4475
|
-
* How many recipe variants that fit this rule's settings have not yet been
|
|
4476
|
-
* tried on the conversations the pipeline has now. Not capped by the free
|
|
4477
|
-
* version slots; 0 while exploration is off or when no variant fits. A count
|
|
4478
|
-
* is not a reason: read `recipe_exploration` for whether one can run next.
|
|
4479
|
-
*/
|
|
4480
|
-
recipe_variants_left: number;
|
|
4481
|
-
/** Whether the next recipe variant can be tried, and if not, why. */
|
|
4482
|
-
recipe_exploration: PipelineRecipeExploration;
|
|
4483
3830
|
/** The rule's monthly limit. Null means it has none. */
|
|
4484
3831
|
monthly_ceiling_cents: number | null;
|
|
4485
3832
|
/**
|
|
@@ -4496,24 +3843,10 @@ export interface Pipeline {
|
|
|
4496
3843
|
*/
|
|
4497
3844
|
month_spent_cents: number;
|
|
4498
3845
|
active_run: PipelineActiveRun | null;
|
|
3846
|
+
/** Winners only, v1 first. */
|
|
4499
3847
|
versions: PipelineVersion[];
|
|
4500
|
-
|
|
4501
|
-
|
|
4502
|
-
rules: TrainingRule[];
|
|
4503
|
-
total: number;
|
|
4504
|
-
}
|
|
4505
|
-
export interface TrainingRuleResponse {
|
|
4506
|
-
rule: TrainingRule;
|
|
4507
|
-
recent_runs: TrainingRunSummary[];
|
|
4508
|
-
/** What this rule's runs have cost at most this UTC calendar month; see {@link Pipeline.month_spent_cents}. */
|
|
4509
|
-
month_spent_cents: number;
|
|
4510
|
-
judge_agreement: JudgeAgreement | null;
|
|
4511
|
-
}
|
|
4512
|
-
export interface TrainingRuleMutationResponse {
|
|
4513
|
-
rule: TrainingRule;
|
|
4514
|
-
/** The rule will not fire again until someone confirms the new amounts. */
|
|
4515
|
-
consent_required: boolean;
|
|
4516
|
-
preflight: TrainingRulePreflight | null;
|
|
3848
|
+
/** Every attempt, attempt 1 first. */
|
|
3849
|
+
attempts: PipelineAttempt[];
|
|
4517
3850
|
}
|
|
4518
3851
|
export interface TrainingRuleDeleteResponse {
|
|
4519
3852
|
deleted: boolean;
|
|
@@ -4552,9 +3885,6 @@ export interface EvaluationItemsResponse {
|
|
|
4552
3885
|
items: EvaluationItem[];
|
|
4553
3886
|
total: number;
|
|
4554
3887
|
}
|
|
4555
|
-
export interface AgentSettingsResponse {
|
|
4556
|
-
settings: AgentSettings;
|
|
4557
|
-
}
|
|
4558
3888
|
export interface InferenceAliasListResponse {
|
|
4559
3889
|
aliases: InferenceAlias[];
|
|
4560
3890
|
}
|
|
@@ -4578,14 +3908,234 @@ export interface BenchmarkHistoryResponse {
|
|
|
4578
3908
|
points: BenchmarkHistoryPoint[];
|
|
4579
3909
|
total: number;
|
|
4580
3910
|
}
|
|
3911
|
+
export interface PipelineListParams {
|
|
3912
|
+
/** How many of the newest pipelines to skip: the page after one that started at N starts at N + its length. */
|
|
3913
|
+
offset?: number;
|
|
3914
|
+
}
|
|
4581
3915
|
export interface PipelineListResponse {
|
|
4582
3916
|
/** The newest pipelines first, at most the service's page (100). */
|
|
4583
3917
|
pipelines: Pipeline[];
|
|
4584
3918
|
/** Every pipeline in the workspace, including any not returned. */
|
|
4585
3919
|
total: number;
|
|
4586
|
-
/** More pipelines exist
|
|
3920
|
+
/** More pipelines exist after this page. */
|
|
4587
3921
|
truncated: boolean;
|
|
4588
3922
|
}
|
|
4589
3923
|
export interface PipelineResponse {
|
|
4590
3924
|
pipeline: Pipeline;
|
|
4591
3925
|
}
|
|
3926
|
+
/**
|
|
3927
|
+
* The body of `POST /api/loop/pipelines` and (every field optional)
|
|
3928
|
+
* `PUT /api/loop/pipelines/{id}`. Everything else -- the machines, the money,
|
|
3929
|
+
* the adapter, the held-back set, the consent -- the platform decides; a field
|
|
3930
|
+
* not listed here is refused by name. See the loop-service README, "Simple
|
|
3931
|
+
* pipelines".
|
|
3932
|
+
*/
|
|
3933
|
+
export interface PipelineWriteRequest {
|
|
3934
|
+
workspace_id?: string;
|
|
3935
|
+
/** Required on a create. The pipeline's own tag is this name made into a tag. */
|
|
3936
|
+
name?: string;
|
|
3937
|
+
use_case?: string;
|
|
3938
|
+
/** A model from {@link LoopModel}, by id. The server pins its revision. */
|
|
3939
|
+
model?: {
|
|
3940
|
+
id: string;
|
|
3941
|
+
};
|
|
3942
|
+
/** A second model to train on the same data; null removes it. */
|
|
3943
|
+
challenger_model?: {
|
|
3944
|
+
id: string;
|
|
3945
|
+
} | null;
|
|
3946
|
+
/** More tags whose conversations the pipeline trains on. On an edit the list replaces the old one. */
|
|
3947
|
+
tags?: string[];
|
|
3948
|
+
/** At least `min_samples` new samples (250 and up for SFT) AND, when set, the schedule. */
|
|
3949
|
+
train_when?: {
|
|
3950
|
+
min_samples?: number;
|
|
3951
|
+
schedule?: {
|
|
3952
|
+
every: 'day' | 'week';
|
|
3953
|
+
weekday?: number;
|
|
3954
|
+
hour_utc: number;
|
|
3955
|
+
} | null;
|
|
3956
|
+
};
|
|
3957
|
+
promotion?: {
|
|
3958
|
+
policy: PipelinePromotion['policy'];
|
|
3959
|
+
k?: number;
|
|
3960
|
+
};
|
|
3961
|
+
/** 1 to 100, or null for no limit (the default). Counts versions, not attempts. */
|
|
3962
|
+
max_versions?: number | null;
|
|
3963
|
+
enabled?: boolean;
|
|
3964
|
+
/** An edit's compare-and-set. */
|
|
3965
|
+
expected_revision?: number;
|
|
3966
|
+
}
|
|
3967
|
+
/** What a pipeline create, edit or train-now answers. */
|
|
3968
|
+
export interface PipelineMutationResponse {
|
|
3969
|
+
pipeline: Pipeline;
|
|
3970
|
+
}
|
|
3971
|
+
/** One base model a pipeline can train (`GET /api/loop/models`). */
|
|
3972
|
+
export interface LoopModel {
|
|
3973
|
+
/** The repository id a pipeline body names in `model.id`. */
|
|
3974
|
+
id: string;
|
|
3975
|
+
name: string;
|
|
3976
|
+
author: string;
|
|
3977
|
+
family: string;
|
|
3978
|
+
/** Total parameters, in billions. */
|
|
3979
|
+
params_b: number;
|
|
3980
|
+
/** Verified end to end on the platform for training and serving. */
|
|
3981
|
+
recommended: boolean;
|
|
3982
|
+
}
|
|
3983
|
+
/** The floors a new pipeline is held to, in samples (`GET /api/loop/models`). */
|
|
3984
|
+
export interface PipelineFloors {
|
|
3985
|
+
/** Samples a new pipeline's first attempt needs: training rows plus the held-back set. */
|
|
3986
|
+
first_version_samples: number;
|
|
3987
|
+
/** `train_when.min_samples`' default and floor. */
|
|
3988
|
+
new_samples: number;
|
|
3989
|
+
}
|
|
3990
|
+
export interface LoopModelsResponse {
|
|
3991
|
+
models: LoopModel[];
|
|
3992
|
+
floors: PipelineFloors;
|
|
3993
|
+
}
|
|
3994
|
+
/** The promotion rule as the API reads it back. */
|
|
3995
|
+
export interface PromotionPolicyState {
|
|
3996
|
+
policy: string;
|
|
3997
|
+
/** Under k_of_n, how many measurements must improve; null otherwise. */
|
|
3998
|
+
k: number | null;
|
|
3999
|
+
updated_at: string | null;
|
|
4000
|
+
updated_by: string | null;
|
|
4001
|
+
}
|
|
4002
|
+
/** One benchmark on a pipeline's list, attached now or once. */
|
|
4003
|
+
export interface RuleBenchmark {
|
|
4004
|
+
benchmark_id: string;
|
|
4005
|
+
name: string;
|
|
4006
|
+
benchmark_status: string;
|
|
4007
|
+
item_count: number;
|
|
4008
|
+
items_digest: string;
|
|
4009
|
+
/** False for a benchmark this pipeline stopped using; its history stays. */
|
|
4010
|
+
attached: boolean;
|
|
4011
|
+
/** 1 is the primary; null once removed. */
|
|
4012
|
+
priority: number | null;
|
|
4013
|
+
primary: boolean;
|
|
4014
|
+
added_at: string;
|
|
4015
|
+
added_by: string | null;
|
|
4016
|
+
removed_at: string | null;
|
|
4017
|
+
removed_by: string | null;
|
|
4018
|
+
}
|
|
4019
|
+
/** What every `/api/loop/pipelines/{id}/benchmarks` route answers. */
|
|
4020
|
+
export interface PipelineBenchmarksResponse {
|
|
4021
|
+
rule_id: string;
|
|
4022
|
+
promotion_policy: PromotionPolicyState;
|
|
4023
|
+
/** Attached ones in priority order, then removed ones, newest first. */
|
|
4024
|
+
benchmarks: RuleBenchmark[];
|
|
4025
|
+
}
|
|
4026
|
+
/** `POST /api/loop/pipelines/{id}/benchmarks`. */
|
|
4027
|
+
export interface PipelineBenchmarkAttachParams {
|
|
4028
|
+
benchmark_id: string;
|
|
4029
|
+
/** Where it goes on the list, 1 first. Absent puts it last. */
|
|
4030
|
+
priority?: number;
|
|
4031
|
+
}
|
|
4032
|
+
/** `PUT /api/loop/pipelines/{id}/benchmarks/{benchmark_id}`. */
|
|
4033
|
+
export interface PipelineBenchmarkMoveParams {
|
|
4034
|
+
/** 1 makes it the primary. */
|
|
4035
|
+
priority: number;
|
|
4036
|
+
}
|
|
4037
|
+
/** A benchmark any version of this pipeline was, or is now, measured on. */
|
|
4038
|
+
export interface CompareBenchmark {
|
|
4039
|
+
benchmark_id: string;
|
|
4040
|
+
name: string;
|
|
4041
|
+
benchmark_status: string;
|
|
4042
|
+
items_digest: string;
|
|
4043
|
+
attached: boolean;
|
|
4044
|
+
priority: number | null;
|
|
4045
|
+
added_at: string | null;
|
|
4046
|
+
removed_at: string | null;
|
|
4047
|
+
}
|
|
4048
|
+
/** One version on one benchmark. `why` says why there is no score. */
|
|
4049
|
+
export interface VersionBenchmarkScore {
|
|
4050
|
+
benchmark_id: string;
|
|
4051
|
+
/** `scored`, `no_score`, or `not_measured` (the benchmark was added after this version). */
|
|
4052
|
+
state: string;
|
|
4053
|
+
benchmark_run_id: string | null;
|
|
4054
|
+
replay_status: string | null;
|
|
4055
|
+
candidate_score: number | null;
|
|
4056
|
+
incumbent_score: number | null;
|
|
4057
|
+
score_delta: number | null;
|
|
4058
|
+
rows_scored: number | null;
|
|
4059
|
+
rows_total: number | null;
|
|
4060
|
+
items_digest: string | null;
|
|
4061
|
+
comparable: boolean;
|
|
4062
|
+
why: string | null;
|
|
4063
|
+
}
|
|
4064
|
+
/** One measurement a promotion rule weighed. */
|
|
4065
|
+
export interface PromotionMeasurement {
|
|
4066
|
+
kind: string;
|
|
4067
|
+
benchmark_id: string | null;
|
|
4068
|
+
benchmark_name: string | null;
|
|
4069
|
+
benchmark_run_id: string | null;
|
|
4070
|
+
items_digest: string | null;
|
|
4071
|
+
priority: number | null;
|
|
4072
|
+
weight: number;
|
|
4073
|
+
delta: number | null;
|
|
4074
|
+
outcome: string;
|
|
4075
|
+
reason: string;
|
|
4076
|
+
}
|
|
4077
|
+
/** What a version's promotion rule decided, and on which numbers. */
|
|
4078
|
+
export interface PromotionOutcome {
|
|
4079
|
+
policy: string;
|
|
4080
|
+
k: number | null;
|
|
4081
|
+
verdict: string;
|
|
4082
|
+
decided_by: string;
|
|
4083
|
+
reason: string;
|
|
4084
|
+
auto_promote_allowed: boolean;
|
|
4085
|
+
weighted_delta: number | null;
|
|
4086
|
+
measurements: PromotionMeasurement[];
|
|
4087
|
+
}
|
|
4088
|
+
/** One version with every measurement it has. */
|
|
4089
|
+
export interface VersionScores {
|
|
4090
|
+
version: number;
|
|
4091
|
+
run_id: string;
|
|
4092
|
+
/** The attempt that became this version, and the base model it trained. */
|
|
4093
|
+
attempt: number | null;
|
|
4094
|
+
base_model_id: string | null;
|
|
4095
|
+
model_family: string | null;
|
|
4096
|
+
state: string;
|
|
4097
|
+
verdict: string | null;
|
|
4098
|
+
decision: string | null;
|
|
4099
|
+
compared_at: string | null;
|
|
4100
|
+
holdout: {
|
|
4101
|
+
rows_scored: number | null;
|
|
4102
|
+
win_rate: number | null;
|
|
4103
|
+
mean_delta: number | null;
|
|
4104
|
+
};
|
|
4105
|
+
/** One entry per entry of the response's `benchmarks`, in the same order. */
|
|
4106
|
+
benchmarks: VersionBenchmarkScore[];
|
|
4107
|
+
promotion: PromotionOutcome | null;
|
|
4108
|
+
}
|
|
4109
|
+
/** Version b against version a on one benchmark. `delta` is b minus a. */
|
|
4110
|
+
export interface VersionScoreDifference {
|
|
4111
|
+
benchmark_id: string;
|
|
4112
|
+
a_score: number | null;
|
|
4113
|
+
b_score: number | null;
|
|
4114
|
+
delta: number | null;
|
|
4115
|
+
comparable: boolean;
|
|
4116
|
+
why: string | null;
|
|
4117
|
+
}
|
|
4118
|
+
/** `GET /api/loop/pipelines/{id}/versions/compare`. */
|
|
4119
|
+
export interface VersionCompareResponse {
|
|
4120
|
+
rule_id: string;
|
|
4121
|
+
promotion_policy: PromotionPolicyState;
|
|
4122
|
+
benchmarks: CompareBenchmark[];
|
|
4123
|
+
versions: VersionScores[];
|
|
4124
|
+
a: number | null;
|
|
4125
|
+
b: number | null;
|
|
4126
|
+
differences: VersionScoreDifference[];
|
|
4127
|
+
}
|
|
4128
|
+
/** Which two versions to set side by side, by VERSION number (v1 is 1). */
|
|
4129
|
+
export interface VersionCompareParams {
|
|
4130
|
+
a?: number;
|
|
4131
|
+
b?: number;
|
|
4132
|
+
}
|
|
4133
|
+
/**
|
|
4134
|
+
* `POST /api/loop/pipelines/{id}/import`: the import body, `source` optional.
|
|
4135
|
+
* `default_feedback: 'good'` records every answered row without a verdict of
|
|
4136
|
+
* its own as a good example to learn from.
|
|
4137
|
+
*/
|
|
4138
|
+
export type PipelineImportParams = Omit<LoopImportParams, 'source'> & {
|
|
4139
|
+
source?: string;
|
|
4140
|
+
default_feedback?: 'good';
|
|
4141
|
+
};
|