runbios-sdk 0.2.17 → 0.2.18-dev.271

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/dist/types.d.ts CHANGED
@@ -8,13 +8,13 @@ export interface BiOSConfig {
8
8
  orgId?: string;
9
9
  /** Workspace ID. Optional when using API keys (resolved from the key). Can override for multi-workspace keys. */
10
10
  workspaceId?: string;
11
- /** Base URL for the API. Falls back to the RUNBIOS_BASE_URL environment variable (legacy: BIOS_BASE_URL), then the canonical https://api.runbios.ai hostname. */
11
+ /** Base URL for the API. Falls back to the RUNBIOS_BASE_URL environment variable (legacy: BIOS_BASE_URL), then the canonical https://api-dev.runbios.ai hostname. */
12
12
  baseUrl?: string;
13
13
  /** Request timeout in milliseconds. Defaults to 30000. */
14
14
  timeout?: number;
15
15
  /** Default per-deployment inference key. Can be overridden per inference call. */
16
16
  inferenceKey?: string;
17
- /** Inference base URL. Defaults to baseUrl, then https://api.runbios.ai. */
17
+ /** Inference base URL. Defaults to baseUrl, then https://api-dev.runbios.ai. */
18
18
  inferenceBaseUrl?: string;
19
19
  /** End-to-end inference timeout in milliseconds. Defaults to 15 minutes. */
20
20
  inferenceTimeout?: number;
@@ -461,21 +461,6 @@ export interface DatasetRegisterHFParams {
461
461
  columnMapping?: Record<string, DatasetTrainingField>;
462
462
  importMode?: 'auto' | 'reference' | 'materialize';
463
463
  }
464
- /** Parameters for registering a curated Conscious Loop set as a dataset. */
465
- export interface DatasetRegisterLoopParams {
466
- /** The Conscious Loop dataset to register. */
467
- loopDatasetId: string;
468
- /** Defaults to `loop-<first 8 of the loop id>`. */
469
- name?: string;
470
- /**
471
- * Which rows to take. Defaults to `all`, because the loop already split the
472
- * set by TIME and re-splitting a deliberate split throws that decision away.
473
- * Register `train` and `holdout` separately to score a training job against
474
- * the loop's own held-back rows; both can exist at once.
475
- */
476
- split?: 'all' | 'train' | 'holdout';
477
- workspaceId?: string;
478
- }
479
464
  /** Parameters for searching HuggingFace Hub datasets. */
480
465
  export interface DatasetHubSearchParams {
481
466
  /** Search query. */
@@ -2344,7 +2329,12 @@ export interface LoopMessage {
2344
2329
  name?: string;
2345
2330
  }
2346
2331
  export interface LoopCaptureParams {
2347
- /** The source being recorded. Must be enabled first, or nothing is stored. */
2332
+ /**
2333
+ * The source being recorded. Must be switched on, or nothing is stored. A
2334
+ * pipeline's own tag (`pipeline.own_tag`) is a source its create switched
2335
+ * on, stamping that tag, so sending there reaches the pipeline with no
2336
+ * further setup; any other source is switched on with `setConfig`.
2337
+ */
2348
2338
  deployment_id: string;
2349
2339
  model: string;
2350
2340
  messages: LoopMessage[];
@@ -2362,6 +2352,12 @@ export interface LoopCaptureParams {
2362
2352
  completion_tokens?: number;
2363
2353
  latency_ms?: number;
2364
2354
  metadata?: Record<string, unknown>;
2355
+ /**
2356
+ * Tags for this conversation, on top of the ones its source stamps. A
2357
+ * pipeline trains on every sample carrying any of its tags, so this is how
2358
+ * one conversation reaches another pipeline too. At most 32 in all.
2359
+ */
2360
+ labels?: string[];
2365
2361
  }
2366
2362
  export interface LoopCaptureResult {
2367
2363
  captured: boolean;
@@ -2369,7 +2365,6 @@ export interface LoopCaptureResult {
2369
2365
  trace_id?: string;
2370
2366
  /** Set when something was removed before the record was written. */
2371
2367
  redacted?: boolean;
2372
- download_url?: string;
2373
2368
  /** Why nothing was stored, e.g. `capture_disabled`, `not_sampled`. */
2374
2369
  reason?: string;
2375
2370
  }
@@ -2554,8 +2549,8 @@ export interface LoopSignalParams {
2554
2549
  /**
2555
2550
  * Which training pipeline this feedback feeds, named in the SAME call.
2556
2551
  *
2557
- * A build rule selects conversations by label and a training rule trains from
2558
- * that build rule, so a label IS the pipeline a conversation goes down.
2552
+ * A pipeline trains on the conversations carrying its tags, so a label IS
2553
+ * the pipeline a conversation goes down.
2559
2554
  * Putting one on used to need a second request after this one — and that
2560
2555
  * second request is the one that gets skipped, by a script that handles the
2561
2556
  * 201 and moves on, by a retry that succeeds here and fails there, by an
@@ -2622,6 +2617,18 @@ export interface LoopTraceListParams {
2622
2617
  model?: string;
2623
2618
  /** `captured` (our model's own behaviour) or `imported` (a file somebody brought). */
2624
2619
  origin?: 'captured' | 'imported';
2620
+ /**
2621
+ * Only the conversations one pipeline's selection covers -- any of its tags
2622
+ * -- reviewed or not. The other filters narrow it further.
2623
+ */
2624
+ pipeline?: string;
2625
+ /**
2626
+ * Only the conversations whose deciding verdict is this feedback: `good`
2627
+ * (accepted), `bad` (rejected) or `corrected` (edited, or gold).
2628
+ */
2629
+ signal?: 'good' | 'bad' | 'corrected';
2630
+ /** Only the conversations nobody has given feedback on yet. */
2631
+ unsignalled?: boolean;
2625
2632
  }
2626
2633
  export interface LoopTraceListResponse {
2627
2634
  traces: LoopTrace[];
@@ -2629,69 +2636,6 @@ export interface LoopTraceListResponse {
2629
2636
  limit: number;
2630
2637
  offset: number;
2631
2638
  }
2632
- export interface LoopDatasetCreateParams {
2633
- name: string;
2634
- method: LoopMethod;
2635
- deployment_id?: string;
2636
- from?: string;
2637
- to?: string;
2638
- /** Narrow to one slice. A label with children selects them too. */
2639
- label?: string;
2640
- /**
2641
- * Take this many of what the filters matched. Which ones is arbitrary but
2642
- * REPEATABLE: the same conversations always give the same slice, so two
2643
- * builds of one spec describe the same set.
2644
- */
2645
- sample?: number;
2646
- /** Holds back your most recent work, not a random slice. 0-50. */
2647
- holdout_percent?: number;
2648
- max_items?: number;
2649
- /**
2650
- * Named dimensions, ANDed with each other and with `label`:
2651
- * `{ category: 'billing', language: 'es' }`. This is what turns one captured
2652
- * corpus into a different dataset for every task somebody trains for.
2653
- */
2654
- attributes?: Record<string, string>;
2655
- /** Only the conversations nobody has described yet. */
2656
- unlabelled?: boolean;
2657
- /** Only what ONE model answered — the filter distillation is made of. */
2658
- model?: string;
2659
- /** `captured` (our model's own behaviour) or `imported` (a file somebody brought). */
2660
- origin?: 'captured' | 'imported';
2661
- }
2662
- export interface LoopDataset {
2663
- id: string;
2664
- workspace_id: string;
2665
- name: string;
2666
- method: LoopMethod;
2667
- status: string;
2668
- spec: Record<string, unknown>;
2669
- item_count: number;
2670
- considered_count: number;
2671
- /** Why rows were left out, by reason. */
2672
- rejected_counts: Record<string, number>;
2673
- holdout_count: number;
2674
- holdout_cutoff: string | null;
2675
- created_by: string | null;
2676
- created_at: string;
2677
- completed_at: string | null;
2678
- download_url: string;
2679
- }
2680
- export interface LoopDatasetListParams {
2681
- method?: LoopMethod;
2682
- limit?: number;
2683
- offset?: number;
2684
- }
2685
- export interface LoopDatasetItem {
2686
- id: number;
2687
- trace_id: string;
2688
- split: 'train' | 'holdout';
2689
- payload: Record<string, unknown>;
2690
- dedup_key: string;
2691
- created_at: string;
2692
- /** The conversation this row was built from. */
2693
- trace_url: string;
2694
- }
2695
2639
  export interface LoopConfig {
2696
2640
  deployment_id: string;
2697
2641
  workspace_id: string;
@@ -2701,42 +2645,8 @@ export interface LoopConfig {
2701
2645
  enabled_by: string | null;
2702
2646
  enabled_at: string | null;
2703
2647
  updated_at: string;
2704
- auto_grade: boolean;
2705
- }
2706
- export interface LoopBuildRuleParams {
2707
- name: string;
2708
- method: string;
2709
- /**
2710
- * How many rows reviewed SINCE THE LAST BUILD must exist before this fires
2711
- * again. Minimum 100, and 100 when absent. Counting the whole corpus instead
2712
- * would fire the rule every interval forever, because a total that has
2713
- * crossed a threshold stays across it.
2714
- */
2715
- min_new_rows?: number;
2716
- enabled?: boolean;
2717
- /** The same selection `createDataset` takes, replayed verbatim. */
2718
- spec?: LoopDatasetCreateParams;
2719
- }
2720
- export interface LoopBuildRule {
2721
- id: string;
2722
- workspace_id: string;
2723
- name: string;
2724
- method: string;
2725
- spec: Record<string, unknown>;
2726
- min_new_rows: number;
2727
- enabled: boolean;
2728
- last_built_at: string | null;
2729
- last_dataset_id: string | null;
2730
- /**
2731
- * Why the rule did not fire last time it was checked. A rule quiet because
2732
- * it is waiting looks exactly like one quiet because it is broken; this is
2733
- * the difference.
2734
- */
2735
- last_reason: string | null;
2736
- last_checked_at: string | null;
2737
- created_by: string | null;
2738
- created_at: string;
2739
- updated_at: string;
2648
+ /** Tags stamped, server side, on every conversation captured from this source. */
2649
+ tags: string[];
2740
2650
  }
2741
2651
  export interface LoopConfigParams {
2742
2652
  enabled: boolean;
@@ -2744,365 +2654,109 @@ export interface LoopConfigParams {
2744
2654
  /** Deterministic per conversation, so turns are never split. 0 < rate <= 1. */
2745
2655
  sample_rate?: number;
2746
2656
  /**
2747
- * Continuous rule-based scoring for this source. OMITTED MEANS UNCHANGED:
2748
- * a caller saving retention must not switch grading off for a workspace
2749
- * that turned it on.
2750
- */
2751
- auto_grade?: boolean;
2752
- }
2753
- /** One thing a judge scores, separately from the others. */
2754
- export interface LoopJudgeDimension {
2755
- key: string;
2756
- description?: string;
2757
- }
2758
- /** Which slice of the corpus a judge is responsible for. */
2759
- export interface LoopJudgeSelection {
2760
- label?: string;
2761
- deployment_id?: string;
2762
- from?: string;
2763
- to?: string;
2764
- sample?: number;
2765
- /** Skip conversations that already carry a judge verdict. */
2766
- only_unscored?: boolean;
2767
- }
2768
- export interface LoopJudge {
2769
- id: string;
2770
- workspace_id: string;
2771
- name: string;
2772
- instructions: string;
2773
- dimensions: LoopJudgeDimension[];
2774
- selection: LoopJudgeSelection;
2775
- model: string | null;
2776
- write_gold: boolean;
2777
- enabled: boolean;
2778
- /** The platform's agent runs this judge, on the workspace's own serverless account. */
2779
- auto: boolean;
2780
- /**
2781
- * Was automatic when the agent was turned off. Turning the agent back on
2782
- * restores exactly these judges.
2783
- */
2784
- auto_paused: boolean;
2785
- /**
2786
- * Why the last turn-on did NOT restore this judge. Null is the ordinary
2787
- * state. Set when the resume declined to make it automatic because the model
2788
- * it names is not one the serving gateway will route — handing it back would
2789
- * buy a run that fails every conversation. It stays paused, so pointing it at
2790
- * a model that routes and turning the agent on again brings it back.
2791
- */
2792
- auto_pause_reason: string | null;
2793
- created_by: string | null;
2794
- created_at: string;
2795
- updated_at: string;
2796
- }
2797
- export interface LoopJudgeParams {
2798
- name: string;
2799
- instructions: string;
2800
- dimensions: LoopJudgeDimension[];
2801
- selection?: LoopJudgeSelection;
2802
- model?: string;
2803
- /**
2804
- * Let this judge write the answer that should have been given, not just a
2805
- * score. Off by default: a reference answer written by a model and then
2806
- * trained on is distillation, which is a decision you make deliberately.
2807
- */
2808
- write_gold?: boolean;
2809
- enabled?: boolean;
2810
- /**
2811
- * Hand the running of this judge to the platform's agent: new conversations
2812
- * in its slice are scored as they arrive, one model call each, on THIS
2813
- * WORKSPACE'S OWN serverless account (the agent spends through a managed key
2814
- * of yours, "Conscious Loop" in your key list). Requires `model`. Off, you
2815
- * run the model yourself with startRun / takeWork / postVerdicts.
2816
- */
2817
- auto?: boolean;
2818
- }
2819
- /** One pass of one rubric over one slice. */
2820
- export interface LoopJudgeRun {
2821
- id: string;
2822
- judge_id: string;
2823
- judge_name?: string;
2824
- /**
2825
- * `abandoned` is a run you opened and never drained: nothing was scored on
2826
- * it for a week and items were still waiting. Nothing is deleted and posting
2827
- * verdicts to it still works and still closes it as done. It exists so that
2828
- * `open` keeps meaning "somebody is working on this".
2829
- *
2830
- * `stopped` is a run somebody closed on purpose before it finished, with
2831
- * `stopRun`. Separate from `done` because a run that covered three of forty
2832
- * conversations did not finish its work: every score it recorded is kept
2833
- * either way, and reading one as the other overstates what was evaluated.
2834
- */
2835
- status: 'open' | 'done' | 'abandoned' | 'stopped';
2836
- instructions: string;
2837
- dimensions: LoopJudgeDimension[];
2838
- model: string | null;
2839
- selected: number;
2840
- scored: number;
2841
- failed: number;
2842
- /** Who drains it: your own code ('caller') or the platform's agent ('platform'). */
2843
- runner: 'caller' | 'platform';
2844
- /** The agent's last complaint about this run, or null while it is working. */
2845
- last_error: string | null;
2846
- last_activity_at: string | null;
2847
- created_by: string | null;
2848
- created_at: string;
2849
- finished_at: string | null;
2850
- }
2851
- /** The workspace's managed serverless key the agent spends through. Never the secret. */
2852
- export interface LoopAgentCredential {
2853
- workspace_id: string;
2854
- key_id: string;
2855
- key_prefix: string;
2856
- created_by: string | null;
2857
- created_at: string;
2858
- updated_at: string;
2859
- last_used_at: string | null;
2860
- last_error: string | null;
2861
- revoked_at: string | null;
2862
- /**
2863
- * The most the agent's key may spend on model calls in a calendar month,
2864
- * in cents. `null` = no cap (the default): the agent can spend up to the
2865
- * workspace's serverless balance. Once reached, the agent's model calls
2866
- * are refused until next month and its runs pause with that reason.
2657
+ * Tags stamped on every conversation captured from this source, so live
2658
+ * traffic reaches the pipelines they feed. Omitted means unchanged; [] clears.
2867
2659
  */
2868
- monthly_spend_cap_cents: number | null;
2869
- }
2870
- export interface LoopAgentStatus {
2871
- /** An agent can exist in this environment at all. */
2872
- available: boolean;
2873
- /** Which piece is missing when it cannot. */
2874
- reason: string;
2875
- /** A worker has checked in within the last minute. */
2876
- online: boolean;
2877
- agent: {
2878
- seen_at: string;
2879
- passes: number;
2880
- judge_items: number;
2881
- samples: number;
2882
- } | null;
2883
- credential: LoopAgentCredential | null;
2884
- open_judge_runs: number;
2885
- open_sample_runs: number;
2886
- auto_judges: number;
2887
- }
2888
- export interface LoopSampleSelection extends LoopJudgeSelection {
2889
- /** Skip conversations that already have alternatives. */
2890
- only_unsampled?: boolean;
2891
- }
2892
- export interface LoopSampleRunParams {
2893
- /** The model that writes the alternatives. For distillation, the teacher. */
2894
- model: string;
2895
- /** Alternatives per conversation, 1-8. Default 4. */
2896
- n?: number;
2897
- /** 0-2. Default 0.8; 0 makes every sample the same answer. */
2898
- temperature?: number;
2899
- /** 16-8192. Default 1024. */
2900
- max_tokens?: number;
2901
- /** Score each sample with this judge as it is written. */
2902
- judge_id?: string;
2903
- selection?: LoopSampleSelection;
2904
- }
2905
- /** "Write N alternatives to each conversation in this slice, and score them." */
2906
- export interface LoopSampleRun {
2907
- id: string;
2908
- workspace_id: string;
2909
- model: string;
2910
- n: number;
2911
- temperature: number;
2912
- max_tokens: number;
2913
- judge_id: string | null;
2914
- judge_name: string | null;
2915
- selection: LoopSampleSelection;
2916
- status: 'open' | 'done';
2917
- selected: number;
2918
- done: number;
2919
- failed: number;
2920
- samples: number;
2921
- last_error: string | null;
2922
- last_activity_at: string | null;
2923
- created_by: string | null;
2924
- created_at: string;
2925
- finished_at: string | null;
2926
- }
2927
- /**
2928
- * One conversation to score, already rendered into the prompt to send.
2929
- *
2930
- * Send `prompt` as-is. Assembling it yourself is how two callers end up giving
2931
- * the same rubric different instructions, and two judges given different
2932
- * instructions are not one judge.
2933
- */
2934
- export interface LoopJudgeWorkItem {
2935
- trace_id: string;
2936
- messages: Array<{
2937
- role: string;
2938
- content?: string;
2939
- }>;
2940
- answer: string;
2941
- prompt: string;
2942
- }
2943
- export interface LoopJudgeWork {
2944
- run: LoopJudgeRun;
2945
- items: LoopJudgeWorkItem[];
2946
- remaining: number;
2947
- }
2948
- export type LoopJudgeRunItemStatus = 'pending' | 'scored' | 'failed';
2949
- /** What happened to one conversation in one run. */
2950
- export interface LoopJudgeRunItem {
2951
- trace_id: string;
2952
- status: LoopJudgeRunItemStatus;
2953
- /**
2954
- * Why this one could not be scored, as the caller reported it. This is where
2955
- * a wrong model name or a refused key shows up; null for anything that is
2956
- * not failed.
2957
- */
2958
- error: string | null;
2959
- scored_at: string | null;
2960
- }
2961
- export interface LoopJudgeRunItems {
2962
- run: LoopJudgeRun;
2963
- items: LoopJudgeRunItem[];
2964
- /** True when the page ended before the run did. */
2965
- has_more: boolean;
2966
- /** Where to carry on from. Present only when `has_more`. */
2967
- next_offset?: number;
2968
- }
2969
- /** One scored conversation going back. */
2970
- export interface LoopJudgeVerdict {
2971
- trace_id: string;
2972
- /** One score per dimension the rubric asked for, each between 0 and 1. */
2973
- scores?: Record<string, number>;
2974
- /** Your own combination of them. Left out, the plain mean is used. */
2975
- overall?: number;
2976
- reason?: string;
2977
- /** Only stored if the judge was created with `write_gold`. */
2978
- gold?: string;
2979
- /** Mark an item you could not score, instead of dropping it silently. */
2980
- error?: string;
2981
- }
2982
- export interface LoopJudgeVerdictResult {
2983
- run: LoopJudgeRun;
2984
- recorded: number;
2985
- failed: number;
2986
- rejected: Array<{
2987
- trace_id: string;
2988
- error: string;
2989
- }>;
2660
+ tags?: string[];
2990
2661
  }
2662
+ /** GET /api/loop/stats: counts for the workspace. */
2991
2663
  export interface LoopStats {
2992
2664
  traces: number;
2993
2665
  signalled_traces: number;
2994
2666
  signals: number;
2995
2667
  by_verdict: Record<string, number>;
2996
2668
  by_source: Record<string, number>;
2997
- datasets: number;
2998
- /** Upper bounds: deduplication runs when a set is built. */
2999
- ready: Record<LoopMethod, number>;
3000
- }
3001
- export interface LoopCandidateParams {
3002
- completion?: string;
3003
- /** An alternative that is itself a tool call. */
3004
- tool_calls?: LoopToolCall[];
3005
- /** Which model produced this alternative. The teacher, when distilling. */
3006
- model?: string;
3007
- model_version?: string;
3008
- /** Unscored alternatives cannot pair -- a missing score is not a low one. */
3009
- score?: number;
3010
- /** Required alongside a score: human | verifier | judge | behavioural. */
3011
- score_source?: LoopSource;
3012
- reason?: string;
3013
- metadata?: Record<string, unknown>;
3014
- }
3015
- export interface LoopCandidate {
3016
- id: string;
3017
- trace_id: string;
3018
- model: string | null;
3019
- model_version: string | null;
3020
- completion: string;
3021
- tool_calls?: LoopToolCall[];
3022
- score: number | null;
3023
- score_source: LoopSource | null;
3024
- reason: string | null;
3025
- metadata: Record<string, unknown>;
3026
- created_at: string;
3027
- }
3028
- /**
3029
- * The deterministic checks a grader can perform.
3030
- *
3031
- * `matches_gold` is the one kind with no expected value of its own: it
3032
- * compares each answer to the gold answer recorded on that same conversation
3033
- * (or the human correction when there is no gold), scores sampled
3034
- * alternatives against the same gold, and is NOT APPLIED to a conversation
3035
- * that carries neither, so it never marks down an unlabelled answer.
3036
- */
3037
- export type LoopGraderKind = 'matches_gold' | 'exact_match' | 'contains' | 'regex' | 'json_valid' | 'numeric' | 'tool_called';
3038
- export interface LoopGraderConfig {
3039
- expected?: string;
3040
- /** Every phrase that must appear. */
3041
- required?: string[];
3042
- /** Phrases that must not. On their own these are passed by silence. */
3043
- forbidden?: string[];
3044
- pattern?: string;
3045
- /** Keys the answer must carry, for `json_valid`. */
3046
- keys?: string[];
3047
2669
  /**
3048
- * How far from `expected` still counts, for `numeric`. For `matches_gold`,
3049
- * how far from the gold answer still counts when both are bare numbers.
2670
+ * How many samples a pipeline could train on now: the ones marked good or
2671
+ * corrected. A pipeline trains SFT only, so this is the one count. An upper
2672
+ * bound: duplicates, the held-back set and conversations too long to train
2673
+ * on come out of it when an attempt trains.
3050
2674
  */
3051
- tolerance?: number;
3052
- /** The function that must have been called, for `tool_called`. */
3053
- function?: string;
3054
- case_sensitive?: boolean;
2675
+ ready: {
2676
+ sft: number;
2677
+ };
3055
2678
  }
3056
- export interface LoopGraderParams {
2679
+ /** One pipeline a conversation falls inside (GET /api/loop/traces/{id}/pipelines). */
2680
+ export interface LoopTracePipeline {
2681
+ /** The pipeline's id: `getPipeline(rule_id)`, the Pipeline's own `rule_id`. */
2682
+ rule_id: string;
2683
+ /** The pipeline's name. */
3057
2684
  name: string;
3058
- kind: LoopGraderKind;
3059
- config: LoopGraderConfig;
3060
- /** The multiplier. Twice as important, twice the weight. Must exceed zero. */
3061
- weight?: number;
3062
- enabled?: boolean;
3063
- /** Limit the rule to one capture source. Omit to apply it everywhere. */
3064
- deployment_id?: string;
3065
- /**
3066
- * Aim the rule at one slice of the corpus. A master group covers its
3067
- * children. Outside that slice the rule is NOT APPLIED, rather than failed,
3068
- * so it stays out of the score entirely. Omit to apply it everywhere.
3069
- */
3070
- label?: string;
3071
- }
3072
- export interface LoopGrader extends LoopGraderParams {
3073
- id: string;
3074
- workspace_id: string;
3075
- weight: number;
2685
+ /** Whether the pipeline is switched on. A paused one still selects the conversation. */
3076
2686
  enabled: boolean;
3077
- created_by: string | null;
2687
+ }
2688
+ export interface LoopTracePipelinesResponse {
2689
+ /** Live pipelines whose tags select this conversation. Empty is a real answer. */
2690
+ pipelines: LoopTracePipeline[];
2691
+ /** Being selected is not being trained on; this sentence says so. */
2692
+ note: string;
2693
+ }
2694
+ export interface LoopMetricsHistoryParams {
2695
+ /** One pipeline's id. Absent, every attempt in the workspace. */
2696
+ rule_id?: string;
2697
+ /** Most recent attempts to return: 200 when absent, at most 500. */
2698
+ limit?: number;
2699
+ /** Days of feedback counts: 90 when absent, at most 365. */
2700
+ days?: number;
2701
+ }
2702
+ /** One attempt, and the comparison it produced if it produced one. */
2703
+ export interface LoopMetricPoint {
2704
+ run_id: string;
2705
+ seq: number;
2706
+ rule_id: string | null;
2707
+ rule_name: string | null;
2708
+ state: TrainingRunState;
2709
+ verdict: TrainingVerdict | null;
2710
+ decision: TrainingDecision | null;
3078
2711
  created_at: string;
3079
- updated_at: string;
2712
+ finished_at: string | null;
2713
+ train_rows: number | null;
2714
+ holdout_rows: number | null;
2715
+ billed_training_cents: number;
2716
+ billed_candidate_cents: number;
2717
+ spent_eval_cents: number;
2718
+ benchmark_spent_cents: number;
2719
+ evaluation_id: string | null;
2720
+ rows_scored: number | null;
2721
+ rows_failed: number | null;
2722
+ wins: number | null;
2723
+ losses: number | null;
2724
+ ties: number | null;
2725
+ win_rate: number | null;
2726
+ incumbent_mean: number | null;
2727
+ candidate_mean: number | null;
2728
+ mean_delta: number | null;
2729
+ judge_incumbent_mean: number | null;
2730
+ judge_candidate_mean: number | null;
2731
+ trainer_eval_loss: number | null;
2732
+ /** The bar this attempt had to clear, frozen with its report. */
2733
+ promote_margin: number | null;
2734
+ promote_min_win_rate: number | null;
2735
+ checkpoint_id: string | null;
2736
+ checkpoint_step: number | null;
2737
+ attempt_no: number | null;
2738
+ /** The version this attempt became; null for every attempt that did not. */
2739
+ version_no: number | null;
2740
+ base_model_id: string | null;
2741
+ model_family: string | null;
2742
+ base_model_revision: string | null;
2743
+ train_type: string | null;
2744
+ trigger: string;
2745
+ recipe_note: string | null;
2746
+ recipe_of_version: number | null;
3080
2747
  }
3081
- export interface LoopGraderResult {
3082
- grader_id: string;
3083
- name: string;
3084
- kind: string;
3085
- weight: number;
3086
- passed: boolean;
3087
- score: number;
3088
- /** What happened, in words -- "failed" alone sends you to read the rule. */
3089
- detail: string;
3090
- /** The rule had no opinion. Excluded from the score rather than counted 0. */
3091
- skipped: boolean;
3092
- }
3093
- export interface LoopGradeReport {
3094
- /** Weighted mean over the rules that applied, 0 to 1. */
3095
- score: number;
3096
- results: LoopGraderResult[];
3097
- /** The denominator. 1.0 from one rule is not 1.0 from six. */
3098
- applied: number;
3099
- skipped: number;
3100
- }
3101
- export interface LoopGradeResult {
3102
- report: LoopGradeReport;
3103
- /** Null when no rule applied, because no verdict was invented. */
3104
- signal_id: string | null;
3105
- candidates_graded: number;
2748
+ /** One day of feedback, counted by verdict. The workspace's, whatever rule_id says. */
2749
+ export interface LoopSignalDay {
2750
+ day: string;
2751
+ total: number;
2752
+ by_verdict: Record<string, number>;
2753
+ }
2754
+ export interface LoopMetricsHistory {
2755
+ /** Oldest first, so a chart reads left to right. */
2756
+ runs: LoopMetricPoint[];
2757
+ signals: LoopSignalDay[];
2758
+ /** True when older attempts exist beyond `limit`. */
2759
+ truncated: boolean;
3106
2760
  }
3107
2761
  export interface LoopLabel {
3108
2762
  label: string;
@@ -3144,8 +2798,6 @@ export interface LoopLabelCount {
3144
2798
  traces: number;
3145
2799
  key: string;
3146
2800
  }
3147
- export type TrainingCadence = 'none' | 'daily' | 'weekly';
3148
- export type TrainingCombinator = 'and' | 'or';
3149
2801
  /** What answers the customer's traffic today, and therefore what promotion re-points. */
3150
2802
  export type ServingKind = 'serverless_slug' | 'deployment';
3151
2803
  export type TrainingRulePausedReason = 'consent_invalid' | 'insufficient_funds' | 'authorizer_not_member' | 'model_not_trainable' | 'platform_capacity'
@@ -3157,45 +2809,6 @@ export type TrainingRulePausedReason = 'consent_invalid' | 'insufficient_funds'
3157
2809
  * wrong; the platform team enables training, then a member saves the rule.
3158
2810
  */
3159
2811
  | 'training_not_enabled';
3160
- /**
3161
- * How a training rule trains, on the wire of the automatic-training routes.
3162
- *
3163
- * DELIBERATELY NOT {@link TrainingMethod}. That union -- 'sft' | 'pt' -- is
3164
- * training-service's, and the two services enumerate different things: a
3165
- * training job can be a plain pre-training run, and a training rule cannot,
3166
- * while a rule may ask for preference training and a job asks for that through
3167
- * a different field. The values here are the loop service's own constants
3168
- * (TrainingMethodSFT / TrainingMethodRLHF in
3169
- * services/loop-service/cmd/training_types.go), which is what these routes
3170
- * accept and return. Sharing the training-service union here advertised 'pt',
3171
- * which this service refuses, and made 'rlhf', which it returns, unspellable.
3172
- *
3173
- * `rlhf_type` stays a plain string on purpose: the platform refuses it until
3174
- * capabilities enable preference training, and an SDK union would have to be
3175
- * republished to keep up with a server-side capability flag.
3176
- */
3177
- export type LoopTrainingMethod = 'sft' | 'rlhf';
3178
- /**
3179
- * What a rule's stored `train_type` may READ as.
3180
- *
3181
- * Still three, and deliberately: the column has always allowed `full`, so a
3182
- * rule created before standing rules were narrowed to adapters can still come
3183
- * back carrying it. Narrowing the response type would make this SDK
3184
- * misrepresent a row that really exists.
3185
- */
3186
- export type TrainType = 'lora' | 'qlora' | 'full';
3187
- /**
3188
- * What a rule may be SET to, which is narrower.
3189
- *
3190
- * loop-service refuses `full` on preflight, create and update for a standing
3191
- * rule: a rule trains unattended and on repeat, and a full fine-tune rewrites
3192
- * every weight instead of adding a small adapter, so it cannot be compared or
3193
- * rolled back cheaply. Typing the request as the wider set handed callers a
3194
- * value guaranteed to come back a 400. One-off full fine-tunes are unaffected
3195
- * — they are a different API.
3196
- */
3197
- export type RuleTrainType = 'lora' | 'qlora';
3198
- export type GradersScope = 'source' | 'all' | 'none';
3199
2812
  /**
3200
2813
  * Why a run fired. `variant` is a recipe variant: the pipeline had nothing new
3201
2814
  * to learn, so it trained the same base model on the same conversations with
@@ -3226,7 +2839,6 @@ export type EvaluationItemStatus = 'pending' | 'working' | 'done' | 'failed';
3226
2839
  export type EvaluationWinner = 'candidate' | 'incumbent' | 'tie';
3227
2840
  /** Typed warning codes. Each surface renders its own plain sentence. */
3228
2841
  export type EvaluationWarningCode = 'holdout_too_small' | 'judge_unreliable' | 'graders_not_applied' | 'many_failures' | 'no_judge' | 'judge_labelled_training_rows' | 'capture_source_moved';
3229
- export type ConsentVia = 'console' | 'api_key' | 'sdk' | 'mcp';
3230
2842
  /** A function an assistant turn asked to invoke. `arguments` is JSON-encoded. */
3231
2843
  export interface TrainingToolCall {
3232
2844
  id?: string;
@@ -3251,386 +2863,6 @@ export interface TrainingRuleServing {
3251
2863
  /** Set only when kind is 'deployment'. */
3252
2864
  deployment_id: string | null;
3253
2865
  }
3254
- /** One rung of a ranked GPU ladder. The provider is the neutral public brand. */
3255
- export interface TrainingGPURung {
3256
- gpu_type: string;
3257
- gpu_count: number;
3258
- provider: string;
3259
- region: string;
3260
- tier: string;
3261
- }
3262
- /** The consent object: one standing instruction to train, judge and possibly promote. */
3263
- export interface TrainingRule {
3264
- id: string;
3265
- workspace_id: string;
3266
- build_rule_id: string;
3267
- name: string;
3268
- enabled: boolean;
3269
- /** Why the platform stopped firing this rule. Waiting must not look like broken. */
3270
- paused_reason: TrainingRulePausedReason | null;
3271
- cadence: TrainingCadence;
3272
- cadence_hour_utc: number;
3273
- cadence_weekday: number | null;
3274
- combinator: TrainingCombinator;
3275
- min_new_rows: number | null;
3276
- /** Spreads firings across the hour so every daily rule does not land on one minute. */
3277
- jitter_seconds: number;
3278
- next_due_at: string | null;
3279
- last_checked_at: string | null;
3280
- last_fired_at: string | null;
3281
- last_run_id: string | null;
3282
- /** Rendered verbatim: "waiting: ...", "due, but ...", "fired: run 7 from ...". */
3283
- last_reason: string | null;
3284
- serving: TrainingRuleServing;
3285
- model_id: string;
3286
- /** A 40-hex commit, pinned when the rule is saved. */
3287
- model_revision: string;
3288
- training_method: LoopTrainingMethod;
3289
- /** Refused by the platform until capabilities enable preference training. */
3290
- rlhf_type: string | null;
3291
- train_type: TrainType;
3292
- config: Record<string, unknown>;
3293
- train_gpu_priorities: TrainingGPURung[];
3294
- train_max_price_hour_cents: number;
3295
- deploy_gpu_priorities: TrainingGPURung[];
3296
- deploy_max_price_hour_cents: number;
3297
- training_ceiling_cents: number;
3298
- candidate_ceiling_cents: number;
3299
- /** A dollar figure, never a call count. */
3300
- eval_ceiling_cents: number;
3301
- monthly_ceiling_cents: number | null;
3302
- eval_judge_id: string | null;
3303
- eval_judge_model: string | null;
3304
- eval_graders_scope: GradersScope;
3305
- eval_max_rows: number;
3306
- eval_max_tokens: number;
3307
- min_holdout_rows: number;
3308
- /**
3309
- * How many versions this pipeline may make, 1 to 10 (five unless changed).
3310
- * A version is a trained model whose comparison was reported having scored
3311
- * at least one conversation, or one a member put live; a run that failed,
3312
- * was cancelled or compared nothing is not one and does not count. At the
3313
- * limit the pipeline stops, and a manual run is refused with
3314
- * `409 VERSION_LIMIT_REACHED`, until the limit is raised.
3315
- */
3316
- max_versions: number;
3317
- /**
3318
- * Try other training recipes when there is nothing new to learn.
3319
- *
3320
- * MONEY-BEARING. When on, a pipeline that has made at least one version, and
3321
- * has nothing reviewed since it that adds anything new to train on (nothing
3322
- * reviewed at all, or only reviews -- thumbs-down with no correction, say --
3323
- * that give it no row the last version's set lacked), trains the same base
3324
- * model on the same conversations with ONE setting changed, and that model
3325
- * takes over only if it beats what serves the app, like any other version.
3326
- * Each try is a full paid run inside the rule's per-run ceilings and its
3327
- * monthly limit, so changing this in EITHER direction changes the terms: the
3328
- * rule stops firing until the member accepts them again. False unless
3329
- * somebody turned it on.
3330
- */
3331
- explore_recipes: boolean;
3332
- /**
3333
- * The standing benchmark replayed on every run of this rule, beside the
3334
- * per-run comparison and never instead of it. Null is the ordinary state.
3335
- *
3336
- * Set with `setTrainingRuleBenchmark`, not by an update to the rule: it is a
3337
- * decision to replay a fixed set on every future run, and it has refusals of
3338
- * its own. It raises no amount, so it does not invalidate consent.
3339
- */
3340
- benchmark_id: string | null;
3341
- /**
3342
- * Whether that benchmark DECIDES the verdict, or only reports a number.
3343
- *
3344
- * False for every rule that has not asked, which is the point of it being
3345
- * separate from `benchmark_id`. Attaching a benchmark means "replay this
3346
- * fixed set on every run and put the score on the report"; it does not mean
3347
- * "let that score overrule the comparison this rule was built on".
3348
- *
3349
- * When it is on it cuts both ways. A run whose held-out split came up short
3350
- * can be decided at all -- before this, such a run trained a model, billed a
3351
- * GPU and could never promote -- and a candidate that wins on fresh
3352
- * conversations while losing ground on the fixed set is refused.
3353
- */
3354
- benchmark_decides: boolean;
3355
- /**
3356
- * How many of the benchmark's pinned conversations must have scored before
3357
- * it may decide anything. A replay that got through four of its forty has
3358
- * not measured the new model.
3359
- */
3360
- benchmark_min_rows: number;
3361
- /**
3362
- * The bar on the replay's own candidate-minus-incumbent delta. Like-for-like
3363
- * WITHIN one replay -- both models, same pinned rows, same frozen judge, one
3364
- * pass -- and not comparable between runs.
3365
- */
3366
- benchmark_min_delta: number;
3367
- auto_promote: boolean;
3368
- promote_margin: number;
3369
- promote_min_win_rate: number;
3370
- /** A gate, not a badge. */
3371
- min_judge_agreement: number;
3372
- review_window_hours: number;
3373
- candidate_boot_deadline_minutes: number;
3374
- keep_candidate_warm_minutes: number;
3375
- /** The member whose wallet pays, and the identity every peer call is stamped with. */
3376
- authorized_by: string;
3377
- terms_version: string;
3378
- revision: number;
3379
- /** Equal to revision only while the consent is current. */
3380
- accepted_revision: number;
3381
- accepted_at: string | null;
3382
- deleted_at: string | null;
3383
- created_at: string;
3384
- updated_at: string;
3385
- }
3386
- /** One append-only record of a member agreeing to spend, with the words they read. */
3387
- export interface TrainingRuleConsent {
3388
- id: number;
3389
- rule_id: string;
3390
- workspace_id: string;
3391
- revision: number;
3392
- terms_version: string;
3393
- terms_text: string;
3394
- snapshot: Record<string, unknown>;
3395
- accepted_by: string;
3396
- accepted_via: ConsentVia;
3397
- accepted_at: string;
3398
- }
3399
- export interface TrainingRulePreflightRefusal {
3400
- stage: string;
3401
- code: string;
3402
- message: string;
3403
- }
3404
- /** One key that cannot follow a cutover, named so the caller can say which. */
3405
- export interface TrainingRuleKeyRef {
3406
- key_id: string;
3407
- prefix: string;
3408
- name: string;
3409
- }
3410
- export interface TrainingRuleKeyCheck {
3411
- keys_missing_deployments_read: TrainingRuleKeyRef[];
3412
- }
3413
- /** The side-effect-free estimate, and the exact sentence the member will accept. */
3414
- export interface TrainingRulePreflight {
3415
- valid: boolean;
3416
- model_revision: string;
3417
- worst_hourly_training_cents: number;
3418
- worst_hourly_candidate_cents: number;
3419
- max_training_hours: number;
3420
- max_candidate_hours: number;
3421
- /** Informational. The fence on evaluation is the dollar ceiling, never this. */
3422
- eval_calls_max: number;
3423
- training_ceiling_cents: number;
3424
- candidate_ceiling_cents: number;
3425
- eval_ceiling_cents: number;
3426
- warnings: string[];
3427
- refusals: TrainingRulePreflightRefusal[];
3428
- /**
3429
- * Every peer the platform could not reach on this pass, one entry each, in
3430
- * the same `{stage, code, message}` shape as a refusal.
3431
- *
3432
- * A warning, not a refusal: an unreachable peer judged nothing, so it never
3433
- * refuses a save. It is still why the rule cannot fire — both
3434
- * `worst_hourly_*` come back `0` — so `valid` is `false`. Key a "cannot
3435
- * save" state on this when `refusals` is empty; the same sentences are also
3436
- * in `warnings`, so render one or the other.
3437
- *
3438
- * Key it on THIS, not on `model_revision`. A degraded pass echoes back the
3439
- * `model_revision` the request carried rather than emptying it, so
3440
- * `if (!model_revision)` is false on exactly the passes it was meant to
3441
- * catch. Use `unreachable.length > 0`, or `/^[0-9a-f]{40}$/`.
3442
- */
3443
- unreachable: TrainingRulePreflightRefusal[];
3444
- key_check: TrainingRuleKeyCheck;
3445
- terms_text: string;
3446
- terms_version: string;
3447
- /**
3448
- * The machines the hourly amounts are per hour of: the ladders the platform
3449
- * chose when the request left them empty, or the caller's own echoed back
3450
- * unchanged. An amount per hour means nothing without knowing what it is per
3451
- * hour of, so show these beside the estimate before anyone accepts it.
3452
- */
3453
- train_gpu_priorities: TrainingGPURung[];
3454
- deploy_gpu_priorities: TrainingGPURung[];
3455
- }
3456
- export interface TrainingRuleBuildSpec {
3457
- method: string;
3458
- spec: Record<string, unknown>;
3459
- }
3460
- /**
3461
- * The editable half of a build rule.
3462
- *
3463
- * Deliberately not `LoopBuildRuleParams`: that type requires `name` and
3464
- * `method`, which the PUT route does not accept, so an edit written against it
3465
- * would have to resend two fields the platform ignores and a caller could not
3466
- * tell that changing them did nothing.
3467
- */
3468
- export interface LoopBuildRuleUpdateParams {
3469
- enabled?: boolean;
3470
- /** Rows reviewed SINCE THE LAST BUILD before this fires again. Minimum 100. */
3471
- min_new_rows?: number;
3472
- /** The same selection `createDataset` takes, replayed verbatim. */
3473
- spec?: LoopDatasetCreateParams;
3474
- }
3475
- /**
3476
- * `?` means absent: leave that part of the sentence alone. `null` is only
3477
- * allowed where the column is nullable and clearing it is a real edit, and it
3478
- * is spelled out field by field rather than applied to the whole object.
3479
- */
3480
- export interface TrainingRuleTriggerInput {
3481
- cadence?: TrainingCadence;
3482
- cadence_hour_utc?: number;
3483
- /** null drops the weekday, which is what a weekly rule moved to daily needs. */
3484
- cadence_weekday?: number | null;
3485
- combinator?: TrainingCombinator;
3486
- /**
3487
- * Conversations reviewed since the last version before the rule fires on
3488
- * rows. At least 100. null removes the row floor, leaving the schedule as the
3489
- * only trigger.
3490
- */
3491
- min_new_rows?: number | null;
3492
- /** 1 to 10. Absent leaves it as it is; absent on a create means five. */
3493
- max_versions?: number;
3494
- /**
3495
- * See {@link TrainingRule.explore_recipes}: every variant it allows is a
3496
- * full paid run, so changing it -- on OR off -- is a money-bearing edit that
3497
- * pauses the rule, and it makes no version of any kind, until the new terms
3498
- * are accepted. Turning it off to save money stops the pipeline too, until
3499
- * somebody accepts again. It sits beside `max_versions`
3500
- * because both say when the pipeline makes another version. Absent leaves
3501
- * it as it is; absent on a create means off.
3502
- */
3503
- explore_recipes?: boolean;
3504
- }
3505
- export interface TrainingRuleTrainingInput {
3506
- model_id?: string;
3507
- model_revision?: string;
3508
- training_method?: LoopTrainingMethod;
3509
- /**
3510
- * A plain string, refused by the platform until capabilities enable
3511
- * preference training. null clears it, which is what moving a rule back to
3512
- * plain supervised training means.
3513
- */
3514
- rlhf_type?: string | null;
3515
- train_type?: RuleTrainType;
3516
- config?: Record<string, unknown>;
3517
- train_gpu_priorities?: TrainingGPURung[];
3518
- train_max_price_hour_cents?: number;
3519
- }
3520
- export interface TrainingRuleDeployInput {
3521
- deploy_gpu_priorities?: TrainingGPURung[];
3522
- deploy_max_price_hour_cents?: number;
3523
- context_length?: number;
3524
- quant?: string;
3525
- serving_config?: Record<string, unknown>;
3526
- hf_integration_id?: string;
3527
- }
3528
- export interface TrainingRuleMoneyInput {
3529
- training_ceiling_cents?: number;
3530
- candidate_ceiling_cents?: number;
3531
- eval_ceiling_cents?: number;
3532
- /** null removes the monthly cap. Absent leaves it exactly where it stands. */
3533
- monthly_ceiling_cents?: number | null;
3534
- /**
3535
- * The most held-back conversations one comparison scores, 10 to 500 and
3536
- * never below `min_holdout_rows`. Absent on a create means 100.
3537
- */
3538
- eval_max_rows?: number;
3539
- }
3540
- export interface TrainingRuleEvaluationInput {
3541
- /** null removes the judge from this rule. */
3542
- judge_id?: string | null;
3543
- /** null drops the override, returning to the judge's own advisory model. */
3544
- judge_model?: string | null;
3545
- graders_scope?: GradersScope;
3546
- /**
3547
- * The fewest held-back conversations that may decide a comparison. At least
3548
- * 20, and 20 when absent on a create. A set that would hold back fewer is
3549
- * not built.
3550
- */
3551
- min_holdout_rows?: number;
3552
- eval_max_tokens?: number;
3553
- }
3554
- export interface TrainingRulePromotionInput {
3555
- auto_promote?: boolean;
3556
- promote_margin?: number;
3557
- promote_min_win_rate?: number;
3558
- min_judge_agreement?: number;
3559
- review_window_hours?: number;
3560
- keep_candidate_warm_minutes?: number;
3561
- }
3562
- /**
3563
- * The version of the terms the member read and accepted.
3564
- *
3565
- * Send back the `terms_version` the preflight returned. The SDK deliberately
3566
- * ships no constant for it: a pinned version in a published package goes stale
3567
- * the moment the platform revises the terms, and the value a member consented
3568
- * to has to be the one they were actually shown.
3569
- */
3570
- export interface TrainingRuleAcceptTerms {
3571
- terms_version: string;
3572
- }
3573
- /** The create body without the yes. The preflight route takes exactly this. */
3574
- export interface TrainingRulePreflightRequest {
3575
- workspace_id?: string;
3576
- name?: string;
3577
- /** Exactly one of build_rule_id and build is given. */
3578
- build_rule_id?: string;
3579
- build?: TrainingRuleBuildSpec;
3580
- trigger?: TrainingRuleTriggerInput;
3581
- serving?: TrainingRuleServing;
3582
- training?: TrainingRuleTrainingInput;
3583
- deploy?: TrainingRuleDeployInput;
3584
- money?: TrainingRuleMoneyInput;
3585
- evaluation?: TrainingRuleEvaluationInput;
3586
- promotion?: TrainingRulePromotionInput;
3587
- enabled?: boolean;
3588
- }
3589
- export interface TrainingRuleCreateRequest extends TrainingRulePreflightRequest {
3590
- /** Absent is a refusal, not a default. */
3591
- accept_terms?: TrainingRuleAcceptTerms;
3592
- /**
3593
- * The standing benchmark this rule replays, set here so it applies to the
3594
- * rule's FIRST run.
3595
- *
3596
- * A rule can fire within seconds of being created, and every step of a run
3597
- * reads the benchmark off the rule as it stood at that moment. Attaching one
3598
- * afterwards with `setTrainingRuleBenchmark` applies from the next run and
3599
- * says nothing about the first, which is the run somebody is watching.
3600
- *
3601
- * Refused on the same terms the attach route refuses it: `404` when it is not
3602
- * this workspace's, `409 BENCHMARK_RETIRED` when it has been retired. Use
3603
- * `setTrainingRuleBenchmark` to change or remove it later; it is not on the
3604
- * preflight body and not on the update body.
3605
- */
3606
- benchmark_id?: string;
3607
- /**
3608
- * And whether that benchmark decides, for the same reason `benchmark_id` is
3609
- * here: run 1 reads its gate off the snapshot it freezes, and a gate applied
3610
- * by a second call lands after run 1 has been decided without it.
3611
- *
3612
- * `400 BENCHMARK_GATE_HAS_NO_SET` when it is true with no benchmark,
3613
- * `400 BENCHMARK_GATE_NEEDS_MIN_ROWS` when it is true with no floor.
3614
- */
3615
- benchmark_decides?: boolean;
3616
- /** Cannot exceed the set's size: `400 BENCHMARK_GATE_UNREACHABLE`. */
3617
- benchmark_min_rows?: number;
3618
- /** Zero -- the default -- means "must not lose ground". */
3619
- benchmark_min_delta?: number;
3620
- }
3621
- export interface TrainingRuleUpdateRequest extends TrainingRulePreflightRequest {
3622
- expected_revision?: number;
3623
- accept_terms?: TrainingRuleAcceptTerms;
3624
- }
3625
- export interface TrainingRuleConsentRequest {
3626
- terms_version: string;
3627
- revision: number;
3628
- }
3629
- export interface TrainingRuleListParams {
3630
- enabled?: boolean;
3631
- limit?: number;
3632
- offset?: number;
3633
- }
3634
2866
  export interface TrainingRunPromoteRequest {
3635
2867
  expected_revision?: number;
3636
2868
  /** Required to promote an inconclusive comparison. */
@@ -3706,12 +2938,23 @@ export interface TrainingRun {
3706
2938
  checkpoint_id: string | null;
3707
2939
  checkpoint_step: number | null;
3708
2940
  /**
3709
- * Which version of the pipeline this run produced. Null for a run that is
3710
- * not a version: one that failed, was cancelled, or whose comparison scored
3711
- * nothing, and that nobody put live. `seq` counts attempts; this counts
3712
- * models that competed, plus any a member promoted without a comparison.
2941
+ * Which attempt of its pipeline this run is: 1, 2, 3 ... as runs are made.
2942
+ * Null only for a run from before attempts were numbered that ended before
2943
+ * it competed.
2944
+ */
2945
+ attempt_no: number | null;
2946
+ /**
2947
+ * Which version of the pipeline this run became. Set only when it was put
2948
+ * live -- promoted by the pipeline's rule or by a member -- as the previous
2949
+ * highest plus one; null for every attempt that did not become the live
2950
+ * model. A rollback renumbers nothing.
3713
2951
  */
3714
2952
  version_no: number | null;
2953
+ /** The base model this run trained, and that model's family. */
2954
+ base_model_id: string | null;
2955
+ model_family: string | null;
2956
+ /** Machine time the run held (training job plus comparison machine); null until known. */
2957
+ gpu_seconds: number | null;
3715
2958
  training_eval_loss: number | null;
3716
2959
  candidate_deployment_id: string | null;
3717
2960
  candidate_name: string | null;
@@ -3767,8 +3010,13 @@ export interface TrainingRunSummary {
3767
3010
  error_code: string | null;
3768
3011
  /** See TrainingRun.error_next_step. On the list row too: the Runs table is where a failure is first seen. */
3769
3012
  error_next_step: string | null;
3013
+ /** See TrainingRun.attempt_no. */
3014
+ attempt_no: number | null;
3770
3015
  /** See TrainingRun.version_no. */
3771
3016
  version_no: number | null;
3017
+ /** See TrainingRun.base_model_id and model_family. */
3018
+ base_model_id: string | null;
3019
+ model_family: string | null;
3772
3020
  verdict: TrainingVerdict | null;
3773
3021
  decision: TrainingDecision | null;
3774
3022
  train_rows: number | null;
@@ -3835,15 +3083,6 @@ export interface EvaluationDimensionScore {
3835
3083
  candidate: number;
3836
3084
  delta: number;
3837
3085
  }
3838
- export interface EvaluationGraderScore {
3839
- grader_id: string;
3840
- name: string;
3841
- incumbent: number;
3842
- candidate: number;
3843
- delta: number;
3844
- /** A grader that applied to four rows has not measured anything. */
3845
- items_applied: number;
3846
- }
3847
3086
  export interface EvaluationWarning {
3848
3087
  code: EvaluationWarningCode;
3849
3088
  message: string;
@@ -3896,7 +3135,6 @@ export interface Evaluation {
3896
3135
  judge_instructions: string | null;
3897
3136
  judge_dimensions: EvaluationJudgeDimension[];
3898
3137
  judge_system_prompt: string | null;
3899
- grader_ids: string[];
3900
3138
  decoding: EvaluationDecoding;
3901
3139
  rows_selected: number;
3902
3140
  rows_scored: number;
@@ -3911,11 +3149,7 @@ export interface Evaluation {
3911
3149
  mean_delta: number | null;
3912
3150
  judge_incumbent_mean: number | null;
3913
3151
  judge_candidate_mean: number | null;
3914
- grader_incumbent_mean: number | null;
3915
- grader_candidate_mean: number | null;
3916
- grader_items_applied: number;
3917
3152
  per_dimension: EvaluationDimensionScore[];
3918
- per_grader: EvaluationGraderScore[];
3919
3153
  judge_agreement: JudgeAgreement | null;
3920
3154
  /** Informational: there is no incumbent counterpart to compare it against. */
3921
3155
  trainer_eval_loss: number | null;
@@ -3935,7 +3169,6 @@ export interface Evaluation {
3935
3169
  export interface EvaluationItemSide {
3936
3170
  completion: string | null;
3937
3171
  tool_calls: TrainingToolCall[] | null;
3938
- grader: Record<string, unknown> | null;
3939
3172
  judge: Record<string, unknown> | null;
3940
3173
  score: number | null;
3941
3174
  model_version: string | null;
@@ -3960,34 +3193,18 @@ export interface EvaluationItem {
3960
3193
  status: EvaluationItemStatus;
3961
3194
  error: string | null;
3962
3195
  }
3196
+ /**
3197
+ * Which pairs to read: the ones the attempt won (`candidate`), the ones the
3198
+ * version serving today won (`serving`; stored on the item as `incumbent`),
3199
+ * or the ties.
3200
+ */
3201
+ export type EvaluationItemWinnerFilter = 'candidate' | 'serving' | 'tie';
3963
3202
  export interface EvaluationItemListParams {
3964
- winner?: EvaluationWinner;
3965
- /** Capped at 100 by the service. */
3203
+ winner?: EvaluationItemWinnerFilter;
3204
+ /** 50 when absent; at most 200. */
3966
3205
  limit?: number;
3967
3206
  offset?: number;
3968
3207
  }
3969
- export interface JudgeAgreementParams {
3970
- /** RFC 3339. Narrows the window the agreement is computed over. */
3971
- from?: string;
3972
- to?: string;
3973
- }
3974
- /** Which model, whose words, how much. The cap is pushed to the workspace key. */
3975
- export interface AgentSettings {
3976
- workspace_id: string;
3977
- default_model: string | null;
3978
- judge_system_prompt: string | null;
3979
- sampler_system_prompt: string | null;
3980
- eval_monthly_cap_cents: number | null;
3981
- updated_by: string | null;
3982
- updated_at: string;
3983
- }
3984
- /** Absent leaves a setting alone; a present null returns it to the platform default. */
3985
- export interface AgentSettingsRequest {
3986
- default_model?: string | null;
3987
- judge_system_prompt?: string | null;
3988
- sampler_system_prompt?: string | null;
3989
- eval_monthly_cap_cents?: number | null;
3990
- }
3991
3208
  /** A re-pointable public handle. Promotion is one row write here. */
3992
3209
  export interface InferenceAlias {
3993
3210
  workspace_id: string;
@@ -4012,8 +3229,11 @@ export interface InferenceAliasRequest {
4012
3229
  * after it, and a delete throws the series away.
4013
3230
  */
4014
3231
  export type BenchmarkStatus = 'active' | 'retired';
4015
- /** Where the pinned conversations are copied from. Read once, at creation. */
4016
- export type BenchmarkSourceKind = 'traces' | 'dataset';
3232
+ /**
3233
+ * Where the pinned conversations are copied from. Read once, at creation.
3234
+ * Always `traces`: the conversations named in `trace_ids`.
3235
+ */
3236
+ export type BenchmarkSourceKind = 'traces';
4017
3237
  /**
4018
3238
  * The life of one replay.
4019
3239
  *
@@ -4139,7 +3359,12 @@ export interface BenchmarkRun {
4139
3359
  export interface BenchmarkHistoryPoint {
4140
3360
  benchmark_run_id: string;
4141
3361
  run_id: string;
3362
+ /** Counts every run in the workspace; label points with attempt_no and version. */
4142
3363
  run_seq: number;
3364
+ /** The attempt of its pipeline this point measured. */
3365
+ attempt_no: number | null;
3366
+ /** The version that attempt became; null when it did not become one. */
3367
+ version: number | null;
4143
3368
  rule_id: string | null;
4144
3369
  rule_name: string | null;
4145
3370
  created_at: string;
@@ -4151,12 +3376,15 @@ export interface BenchmarkHistoryPoint {
4151
3376
  status: BenchmarkRunStatus;
4152
3377
  status_reason: string | null;
4153
3378
  }
4154
- /** Where the conversations are copied FROM. Read once, at creation, never again. */
3379
+ /**
3380
+ * Where the conversations are copied FROM. Read once, at creation, never again.
3381
+ * A benchmark pins conversations by id; to make one from a file, import it
3382
+ * (`importRows`, with no tag, so it never trains a pipeline) and pin the
3383
+ * `trace_ids` the import returns.
3384
+ */
4155
3385
  export interface BenchmarkSource {
4156
3386
  kind: BenchmarkSourceKind;
4157
- trace_ids?: string[];
4158
- dataset_id?: string;
4159
- split?: string;
3387
+ trace_ids: string[];
4160
3388
  }
4161
3389
  /**
4162
3390
  * The pin: 20 to 200 conversations, and a judge with a rubric to measure them.
@@ -4165,15 +3393,16 @@ export interface BenchmarkSource {
4165
3393
  * Pinning 47 of the 50 somebody chose is the set being wrong from the first
4166
3394
  * day, and they would never find out.
4167
3395
  */
3396
+ /**
3397
+ * A name and the conversations, nothing else. The platform freezes the judge
3398
+ * and the model it scores on, gives both models the comparison's own answer
3399
+ * length, and sets what one replay may spend; none of that is sent (the
3400
+ * service refuses `judge_id`, `judge_model`, `max_tokens` and
3401
+ * `per_run_ceiling_cents` by name).
3402
+ */
4168
3403
  export interface BenchmarkCreateParams {
4169
3404
  name: string;
4170
- judge_id: string;
4171
3405
  source: BenchmarkSource;
4172
- /** Absent uses the judge's own model, and absent that the workspace default. */
4173
- judge_model?: string;
4174
- /** What both models are given to answer in. A shorter answer is a different answer. */
4175
- max_tokens?: number;
4176
- per_run_ceiling_cents: number;
4177
3406
  }
4178
3407
  export interface BenchmarkListParams {
4179
3408
  status?: BenchmarkStatus;
@@ -4181,11 +3410,12 @@ export interface BenchmarkListParams {
4181
3410
  offset?: number;
4182
3411
  }
4183
3412
  export interface BenchmarkItemListParams {
4184
- /** Capped at 100 by the service. */
3413
+ /** 50 when absent; at most 200. */
4185
3414
  limit?: number;
4186
3415
  offset?: number;
4187
3416
  }
4188
3417
  export interface BenchmarkHistoryParams {
3418
+ /** 50 when absent; at most 200. */
4189
3419
  limit?: number;
4190
3420
  offset?: number;
4191
3421
  }
@@ -4194,77 +3424,96 @@ export interface BenchmarkRetireParams {
4194
3424
  reason?: string;
4195
3425
  }
4196
3426
  /**
4197
- * Attach a benchmark to a rule, detach it with a present null, or move the bar
4198
- * it decides on.
4199
- *
4200
- * ABSENT IS "LEAVE IT", PRESENT IS "MAKE IT THIS", and that applies to every
4201
- * field here. This is the route that attaches a benchmark AND the route that
4202
- * moves its bar a month later, so a call naming only `benchmark_id` must not
4203
- * reset a floor somebody chose, and a call naming only `benchmark_min_delta`
4204
- * must not detach the set.
4205
- *
4206
- * Detaching is the one exception: a present null `benchmark_id` clears
4207
- * `benchmark_decides` with it, because a gate with nothing behind it is a
4208
- * verdict waiting on a measurement that will never come. Asking for both in
4209
- * one request -- null id and `benchmark_decides: true` -- is refused rather
4210
- * than resolved, because either resolution is a guess about which half you
4211
- * meant.
4212
- */
4213
- export interface TrainingRuleBenchmarkRequest {
4214
- benchmark_id?: string | null;
4215
- benchmark_decides?: boolean;
4216
- benchmark_min_rows?: number;
4217
- benchmark_min_delta?: number;
4218
- }
4219
- /**
4220
- * The machine codes `runTrainingRule` can be refused with. Each is the
4221
- * platform declining to start a paid run it would only refuse a minute later,
4222
- * so each is said in front of the caller rather than left for the rule's
4223
- * `last_reason` to explain after they have stopped looking.
3427
+ * The machine codes `runPipeline` (train now) can be refused with: every one
3428
+ * the run route answers (the SDK's tests read them off the service). Each is
3429
+ * the platform declining to start a paid attempt it would only refuse a minute
3430
+ * later, so each is said in front of the caller.
4224
3431
  *
4225
- * - `CONSENT_REQUIRED` (409): the terms changed and nobody accepted them.
4226
- * - `RULE_PAUSED` (409): the rule is switched off or the platform stopped it.
4227
- * - `RUN_ACTIVE` (409): one run at a time, so two cannot race to promote.
3432
+ * - `NOT_ENOUGH_SAMPLES` (422): fewer samples than the next attempt needs.
3433
+ * The body carries `have` and `need`. Train now skips the schedule, never
3434
+ * the minimum.
3435
+ * - `PIPELINE_NEEDS_SAVE` (409): the pipeline's settings changed without a
3436
+ * pipeline save since; save them (saving is the authorization) and train
3437
+ * again.
3438
+ * - `RULE_PAUSED` (409): the pipeline is switched off or the platform paused
3439
+ * it (`paused_reason` says why).
3440
+ * - `RUN_ACTIVE` (409): one attempt at a time, so two cannot race to promote.
4228
3441
  * - `VERSION_LIMIT_REACHED` (409): the pipeline has made `max_versions`
4229
- * versions. Raise the limit or start a new pipeline.
4230
- * - `MONTHLY_LIMIT_REACHED` (409): the most this month's runs can have cost
4231
- * plus the most one more run may cost would pass `monthly_ceiling_cents`.
4232
- * The body carries the figures:
3442
+ * versions. Raise the limit or remove it.
3443
+ * - `MONTHLY_LIMIT_REACHED` (409): the most this month's attempts can have
3444
+ * cost plus the most one more may cost would pass the pipeline's internal
3445
+ * monthly limit. The body carries the figures:
4233
3446
  * see {@link TrainingRuleMonthlyLimitRefusal}.
4234
- * - `AGENT_OFF` (409): the workspace's Conscious Loop agent is not turned on.
4235
- * Every comparison runs under the agent's key, so a run started without one
4236
- * would pay for training and could not be compared. Turn it on in Agent
4237
- * settings, then run the rule.
4238
- * - `NOT_ENOUGH_ROWS` (422): too few reviewed or held-out conversations, with
4239
- * `train_rows`, `holdout_rows` and the `floor` it needed.
4240
- */
4241
- export type TrainingRuleRunRefusalCode = 'CONSENT_REQUIRED' | 'RULE_PAUSED' | 'RUN_ACTIVE' | 'VERSION_LIMIT_REACHED' | 'MONTHLY_LIMIT_REACHED' | 'AGENT_OFF' | 'NOT_ENOUGH_ROWS';
4242
- /**
4243
- * The machine codes a SAVE of a training rule is refused with when the
4244
- * preflight it runs refuses: `createTrainingRule`, an `updateTrainingRule`
4245
- * that switches a rule on or into preference training or carries
4246
- * `accept_terms`, and `consentTrainingRule`. Every other preflight refusal is
4247
- * a `400` carrying the peer's own sentence; `preflightTrainingRule` answers
4248
- * `200` with the same codes in `refusals`.
3447
+ * - `SCORING_CAPPED` (409): the key the comparison runs under is at its
3448
+ * monthly cap, so what this attempt trained could not be scored. The body
3449
+ * carries `resumes_at`, when the cap lifts (the next UTC month, or sooner
3450
+ * if the key's cap is raised).
3451
+ * - `AGENT_COULD_NOT_START` (503): the key every comparison is scored with is
3452
+ * missing or was refused, and train now could not renew it just then.
3453
+ * Nothing was started. Press train now again in a minute.
3454
+ * - `AGENT_OFF` (409): scoring could not run: that key was still unusable
3455
+ * after train now renewed it, so no attempt was started. Press train now
3456
+ * again in a minute. Nothing in the pipeline's settings causes it, so
3457
+ * saving the pipeline does not clear it.
4249
3458
  *
4250
- * - `TRAINING_NOT_ENABLED` (409): training is in beta and this organisation
4251
- * does not hold the grant. Nothing is saved and nothing in the request can
4252
- * lift it, so do not retry: ask the platform team to enable training for
4253
- * the organisation. It is answered ahead of every other refusal. A rule that
4254
- * meets it at fire time is paused `training_not_enabled` instead (see
4255
- * {@link TrainingRulePausedReason}).
4256
- * - `METHOD_NOT_ENABLED` (422): a training method the platform has not enabled.
4257
- * - `CEILING_TOO_LOW` (422): a ceiling cannot buy the minimum hours at the
4258
- * accepted price cap.
3459
+ * Only the last two clear on their own within a minute, and
3460
+ * `MONTHLY_LIMIT_REACHED` and `SCORING_CAPPED` once their date has passed;
3461
+ * retrying any other one unchanged gets the same answer.
4259
3462
  */
4260
- export type TrainingRuleSaveRefusalCode = 'TRAINING_NOT_ENABLED' | 'METHOD_NOT_ENABLED' | 'CEILING_TOO_LOW';
3463
+ export type TrainingRuleRunRefusalCode = 'NOT_ENOUGH_SAMPLES' | 'PIPELINE_NEEDS_SAVE' | 'RULE_PAUSED' | 'RUN_ACTIVE' | 'VERSION_LIMIT_REACHED' | 'MONTHLY_LIMIT_REACHED' | 'SCORING_CAPPED' | 'AGENT_COULD_NOT_START' | 'AGENT_OFF';
4261
3464
  /**
4262
- * The machine codes `deleteDataset` can be refused with.
3465
+ * The machine codes saving a pipeline (`createPipeline` or `updatePipeline`)
3466
+ * can be refused with: every one the two routes answer beyond a malformed
3467
+ * body (400, which names the field). The SDK's tests read them off the
3468
+ * service. Nothing is saved when any of them is answered.
4263
3469
  *
4264
- * - `DATASET_IN_USE` (409): a training run was trained on this set, so it is
4265
- * kept as the record of what that run learned from.
3470
+ * The name and tags:
3471
+ * - `PIPELINE_NAME_TAKEN` (409): another pipeline in the workspace has the
3472
+ * name, a deleted one included. Choose another name.
3473
+ * - `OWN_TAG_TAKEN` (409): another pipeline already owns the tag this name
3474
+ * makes. Choose another name.
3475
+ *
3476
+ * An edit only:
3477
+ * - `REVISION_MISMATCH` (409): `expected_revision` is not the pipeline's
3478
+ * revision any more. Read it again and resend.
3479
+ * - `RUN_ACTIVE` (409): the model cannot change while an attempt is running.
3480
+ * - `NOT_A_SIMPLE_PIPELINE` (409): a tag, use case or challenger edit on a
3481
+ * pipeline created before pipelines had them; its other settings can
3482
+ * still change.
3483
+ *
3484
+ * The model and when to train:
3485
+ * - `MODEL_NOT_TRAINABLE` (422): `model.id` or `challenger_model.id` is not a
3486
+ * model `listModels` offers, or the challenger is the model itself.
3487
+ * - `MIN_SAMPLES_TOO_LOW` (422): `train_when.min_samples` is under the floor
3488
+ * `listModels` returns as `floors.new_samples`.
3489
+ * - `NO_GPU_FITS` (422): no machines can train or serve the model right now.
3490
+ * - `MODEL_REVISION_UNAVAILABLE` (422): the model could not be pinned to an
3491
+ * exact revision because the training service did not answer. Try again
3492
+ * once it does.
3493
+ * - `NO_JUDGE_MODEL` (422): no model the workspace can call is able to score
3494
+ * the attempts, so none could ever be measured.
3495
+ * - `PROMOTION_POLICY_UNKNOWN`, `PROMOTION_K_REQUIRED`,
3496
+ * `PROMOTION_K_UNREACHABLE`, `PROMOTION_POLICY_NEEDS_BENCHMARK` (422): the
3497
+ * promotion policy is not one of the six, `k_of_n` came without `k`, `k`
3498
+ * is more than the measurements the pipeline takes, or the policy needs a
3499
+ * benchmark the pipeline does not have yet.
3500
+ *
3501
+ * The platform:
3502
+ * - `TRAINING_NOT_ENABLED` (409): training is in beta and this organisation
3503
+ * does not hold the grant. Nothing in the request can lift it, so do not
3504
+ * retry: ask the platform team to enable training for the organisation. A
3505
+ * pipeline that meets it at fire time is paused `training_not_enabled`
3506
+ * instead (see {@link TrainingRulePausedReason}).
3507
+ * - `METHOD_NOT_ENABLED` (422): the platform has switched off the training
3508
+ * method pipelines train with.
3509
+ * - `CEILING_TOO_LOW` (422): the platform's own runaway limits cannot cover
3510
+ * the machines it chose for the model. Nothing in the request sets them.
3511
+ * - `MODELS_UNAVAILABLE`, `GPU_OPTIONS_UNAVAILABLE`, `AGENT_COULD_NOT_START`
3512
+ * (503): a service the save needs (the model list, the machine list, or
3513
+ * the key every comparison is scored with) did not answer just then. Try
3514
+ * again in a minute.
4266
3515
  */
4267
- export type LoopDatasetDeleteRefusalCode = 'DATASET_IN_USE';
3516
+ export type TrainingRuleSaveRefusalCode = 'PIPELINE_NAME_TAKEN' | 'OWN_TAG_TAKEN' | 'REVISION_MISMATCH' | 'RUN_ACTIVE' | 'NOT_A_SIMPLE_PIPELINE' | 'MODEL_NOT_TRAINABLE' | 'MIN_SAMPLES_TOO_LOW' | 'NO_GPU_FITS' | 'MODEL_REVISION_UNAVAILABLE' | 'NO_JUDGE_MODEL' | 'PROMOTION_POLICY_UNKNOWN' | 'PROMOTION_K_REQUIRED' | 'PROMOTION_K_UNREACHABLE' | 'PROMOTION_POLICY_NEEDS_BENCHMARK' | 'TRAINING_NOT_ENABLED' | 'METHOD_NOT_ENABLED' | 'CEILING_TOO_LOW' | 'MODELS_UNAVAILABLE' | 'GPU_OPTIONS_UNAVAILABLE' | 'AGENT_COULD_NOT_START';
4268
3517
  /**
4269
3518
  * The body of a manual run refused `409 MONTHLY_LIMIT_REACHED`, as
4270
3519
  * `ApiError.body` carries it.
@@ -4293,7 +3542,7 @@ export interface TrainingRuleMonthlyLimitRefusal {
4293
3542
  resumes_at: string;
4294
3543
  }
4295
3544
  /** The most versions a pipeline may be set to make (loop-service maxVersionsCeiling). */
4296
- export declare const PIPELINE_MAX_VERSIONS_CEILING = 10;
3545
+ export declare const PIPELINE_MAX_VERSIONS_CEILING = 100;
4297
3546
  /**
4298
3547
  * What a pipeline is doing. `needs_review` and `needs_funds` are a finished
4299
3548
  * version waiting on the MEMBER -- a decision, or a top-up -- which is not the
@@ -4305,18 +3554,29 @@ export declare const PIPELINE_MAX_VERSIONS_CEILING = 10;
4305
3554
  */
4306
3555
  export type PipelineStatus = 'running' | 'needs_review' | 'needs_funds' | 'waiting' | 'paused' | 'complete';
4307
3556
  /**
4308
- * The run in flight. It is not a version until its comparison is reported
4309
- * having scored at least one conversation, or a member puts it live.
3557
+ * The attempt in flight. It is not a version until it is put live -- by the
3558
+ * pipeline's promotion rule or by a member.
4310
3559
  */
4311
3560
  export interface PipelineActiveRun {
4312
3561
  run_id: string;
3562
+ /** Its attempt number within the pipeline. */
3563
+ attempt: number | null;
3564
+ /** Why it was made; see {@link PipelineAttempt.trigger}. */
3565
+ trigger: PipelineAttemptTrigger;
3566
+ /** The base model it trains, and that model's family. */
3567
+ base_model_id: string | null;
3568
+ model_family: string | null;
4313
3569
  state: TrainingRunState;
4314
3570
  since: string;
4315
- /**
4316
- * The version number it will take if it becomes one: if its comparison
4317
- * scores something, or if a member promotes it anyway.
4318
- */
3571
+ /** When the attempt was made. */
3572
+ started_at: string;
3573
+ /** Machine time it has held so far; null until known. */
3574
+ gpu_seconds: number | null;
3575
+ /** What it has been billed so far, in cents. */
3576
+ cost_cents: number;
3577
+ /** The version number it will take if it is put live. */
4319
3578
  will_be_version: number;
3579
+ /** Always null: nothing in flight is a version yet. */
4320
3580
  version_no: number | null;
4321
3581
  last_reason: string | null;
4322
3582
  last_error: string | null;
@@ -4332,20 +3592,29 @@ export interface PipelineActiveRun {
4332
3592
  recipe_note: string | null;
4333
3593
  }
4334
3594
  /**
4335
- * One version: a trained model whose comparison scored at least one
4336
- * conversation, or one a member put live without one. The second kind has
4337
- * `rows_scored` 0 and `win_rate` null, and may be the champion.
3595
+ * One version: an attempt that was put live -- promoted by the pipeline's
3596
+ * rule or by a member. Only winners are versions, numbered v1, v2, ... in the
3597
+ * order they went live; numbers only go up and a rollback renumbers nothing.
4338
3598
  */
4339
3599
  export interface PipelineVersion {
4340
3600
  version: number;
4341
3601
  run_id: string;
3602
+ /** The attempt that became this version. */
3603
+ attempt: number | null;
3604
+ /** The base model it was trained from, and that model's family. */
3605
+ base_model_id: string | null;
3606
+ model_family: string | null;
4342
3607
  seq: number;
4343
3608
  state: TrainingRunState;
4344
3609
  verdict: TrainingVerdict | null;
4345
3610
  decision: TrainingDecision | null;
4346
3611
  decided_at: string | null;
3612
+ /** True for the version the app is served by today (the same as `is_champion`). */
3613
+ is_live: boolean;
4347
3614
  /** True for the version the app is served by today. */
4348
3615
  is_champion: boolean;
3616
+ /** When it was first put live. */
3617
+ promoted_at: string | null;
4349
3618
  checkpoint_id: string | null;
4350
3619
  train_rows: number | null;
4351
3620
  holdout_rows: number | null;
@@ -4379,6 +3648,8 @@ export interface PipelineVersion {
4379
3648
  * not what it did.
4380
3649
  */
4381
3650
  cost_includes_candidate_ceiling: boolean;
3651
+ /** Machine time the attempt held (training and comparison); null when unknown. */
3652
+ gpu_seconds: number | null;
4382
3653
  created_at: string;
4383
3654
  /**
4384
3655
  * When this version's head-to-head started. The champion it faced is the
@@ -4403,42 +3674,129 @@ export interface PipelineVersion {
4403
3674
  finished_at: string | null;
4404
3675
  }
4405
3676
  /**
4406
- * A training rule seen as the series of versions it produces: which exist,
4407
- * which one serves, whether each beat the one before it, and what happens
4408
- * next. READ-ONLY: every decision stays on the route that owns it (promote,
4409
- * reject and roll back on the run; the version limit on the rule).
3677
+ * A pipeline: one use case, and the series of versions it produces -- which
3678
+ * exist, which one serves, whether each beat the one before it, and what
3679
+ * happens next. READ-ONLY: every decision stays on the route that owns it
3680
+ * (promote, reject and roll back on the attempt; the version limit is set by
3681
+ * editing the pipeline).
4410
3682
  */
4411
3683
  /**
4412
- * Whether a pipeline can make a recipe variant next, and when it cannot, which
4413
- * reason -- each is a different next step.
4414
- *
4415
- * - `off`: explore_recipes is off.
4416
- * - `waiting_for_v1`: on, but a variant varies a version and there is none yet.
4417
- * - `available`: on, with `recipe_variants_left` untried variants that fit.
4418
- * - `no_slots`: untried variants fit (`recipe_variants_left` > 0), but every
4419
- * version slot is made or taken by the run in flight. Raise `max_versions`
4420
- * (or, at 10, start a new pipeline) and one can run. Only when variants are
4421
- * left: a service that reports `no_slots` with `recipe_variants_left` of 0
4422
- * means `all_tried`, and raising the limit starts nothing.
4423
- * - `all_tried`: every variant that fits has been tried on the conversations
4424
- * it has now, whether or not a version slot is free. Raising `max_versions`
4425
- * does not start one; new reviews bring the next version.
4426
- * - `none_fit`: no variant fits this rule's training settings.
3684
+ * Why an attempt was made: new data (`data`, a recipe variant included), the
3685
+ * schedule (`schedule`), a member's train-now (`manual`), or the challenger
3686
+ * model trained on the same data as the attempt before it (`challenger`).
3687
+ */
3688
+ export type PipelineAttemptTrigger = 'data' | 'schedule' | 'manual' | 'challenger';
3689
+ /** What became of an attempt. */
3690
+ export type PipelineAttemptOutcome = 'became_version' | 'not_better' | 'inconclusive' | 'stopped' | 'failed' | 'running' | 'waiting_for_review';
3691
+ /** One attempt: every training run a pipeline makes, winner or not. */
3692
+ export interface PipelineAttempt {
3693
+ attempt: number;
3694
+ run_id: string;
3695
+ state: TrainingRunState;
3696
+ trigger: PipelineAttemptTrigger;
3697
+ base_model_id: string | null;
3698
+ model_family: string | null;
3699
+ outcome: PipelineAttemptOutcome;
3700
+ /** The version it became, when it became one. */
3701
+ version: number | null;
3702
+ created_at: string;
3703
+ finished_at: string | null;
3704
+ gpu_seconds: number | null;
3705
+ /** As the versions are priced once it has ended; as recorded so far while it runs. */
3706
+ cost_cents: number;
3707
+ verdict: TrainingVerdict | null;
3708
+ win_rate: number | null;
3709
+ evaluation_id: string | null;
3710
+ /** Its last sentence. */
3711
+ reason: string | null;
3712
+ }
3713
+ /** A model and the exact commit of it. */
3714
+ export interface PipelineModelRef {
3715
+ id: string;
3716
+ revision: string;
3717
+ }
3718
+ /**
3719
+ * When a new version takes over. `holdout` (the default) when it wins on the
3720
+ * held-back set; `all`, `primary`, `k_of_n` (with `k`) and `weighted` also
3721
+ * weigh the pipeline's benchmarks; `manual` never on its own.
3722
+ */
3723
+ export interface PipelinePromotion {
3724
+ policy: 'holdout' | 'all' | 'primary' | 'k_of_n' | 'weighted' | 'manual';
3725
+ k: number | null;
3726
+ }
3727
+ /** A pipeline's schedule, UTC: every day at an hour, or every week on a weekday at an hour. */
3728
+ export interface PipelineSchedule {
3729
+ every: 'day' | 'week';
3730
+ /** 0 (Sunday) to 6, for `week`; null for `day`. */
3731
+ weekday: number | null;
3732
+ hour_utc: number;
3733
+ }
3734
+ /**
3735
+ * When a pipeline trains: at least `min_samples` new samples AND, when there
3736
+ * is one, at its scheduled slot. A null schedule trains as soon as the minimum
3737
+ * is met.
3738
+ */
3739
+ export interface PipelineTrainWhen {
3740
+ min_samples: number;
3741
+ schedule: PipelineSchedule | null;
3742
+ }
3743
+ /** What is held back beyond max(50, 5% of the data): a larger percent, or a count. */
3744
+ export interface PipelineHoldout {
3745
+ percent: number;
3746
+ count: number | null;
3747
+ min_rows: number;
3748
+ }
3749
+ /**
3750
+ * How far the data is from the next version. `first_version`: `needed` usable
3751
+ * rows v1 trains on. `new_rows`: `needed` usable rows the last version's set
3752
+ * did not have. `have`, `short` and `message` are measured on the detail only.
4427
3753
  */
4428
- export type PipelineRecipeExploration = 'off' | 'waiting_for_v1' | 'available' | 'no_slots' | 'all_tried' | 'none_fit';
3754
+ export interface PipelineMinimums {
3755
+ basis: 'first_version' | 'new_rows';
3756
+ needed: number;
3757
+ have: number | null;
3758
+ short: number | null;
3759
+ message: string | null;
3760
+ }
4429
3761
  export interface Pipeline {
4430
3762
  rule_id: string;
4431
3763
  rule_name: string;
4432
3764
  enabled: boolean;
4433
- /** The limit the member set, 1 to 10. */
4434
- max_versions: number;
3765
+ /** The settings revision an edit sends back as `expected_revision`. */
3766
+ revision: number;
3767
+ /** The limit the member set, 1 to 100, or null for none. */
3768
+ max_versions: number | null;
4435
3769
  /** Versions made so far. At `max_versions` the pipeline stops. */
4436
3770
  versions_made: number;
3771
+ /** What the pipeline is for, in the member's words. Empty on a pipeline created before pipelines had a use case. */
3772
+ use_case: string;
3773
+ /** The tag imports into this pipeline are given. Null on a pipeline created before pipelines had a tag of their own. */
3774
+ own_tag: string | null;
3775
+ /** A conversation carrying ANY of these tags is this pipeline's data. */
3776
+ tags: string[];
3777
+ /**
3778
+ * A second base model trained on the same data after each attempt; it
3779
+ * becomes the next version only if it beats the live one. `revision` is the
3780
+ * commit the server pinned.
3781
+ */
3782
+ challenger_model: PipelineModelRef | null;
3783
+ /** `sft` today; `kto`, `dpo` and `grpo` cannot be chosen yet. */
3784
+ training_type: string;
3785
+ /** How each version starts: `fresh`, a new adapter over the pinned base trained on all accepted data. */
3786
+ version_base: string;
3787
+ train_when: PipelineTrainWhen;
3788
+ promotion: PipelinePromotion;
3789
+ holdout: PipelineHoldout;
3790
+ /** Rows the next version holds back and never trains on. Detail only; null on the list. */
3791
+ holdout_size: number | null;
3792
+ minimums: PipelineMinimums;
4437
3793
  /** The pinned base every version is a fresh adapter over, which is what makes their scores comparable. */
4438
3794
  base_model_id: string;
4439
3795
  base_model_revision: string;
4440
3796
  train_type: string;
4441
- /** Null when what serves is not a version of this pipeline. */
3797
+ /** The version serving now; null when what serves is not a version of this pipeline. */
3798
+ live_version: number | null;
3799
+ /** The same number under its older name. */
4442
3800
  champion_version: number | null;
4443
3801
  /** The name the app calls. It does not change when a version is promoted. */
4444
3802
  serving_name: string;
@@ -4469,17 +3827,6 @@ export interface Pipeline {
4469
3827
  * until somebody does.
4470
3828
  */
4471
3829
  needs_consent: boolean;
4472
- /** The rule's explore_recipes, as it stands. */
4473
- explore_recipes: boolean;
4474
- /**
4475
- * How many recipe variants that fit this rule's settings have not yet been
4476
- * tried on the conversations the pipeline has now. Not capped by the free
4477
- * version slots; 0 while exploration is off or when no variant fits. A count
4478
- * is not a reason: read `recipe_exploration` for whether one can run next.
4479
- */
4480
- recipe_variants_left: number;
4481
- /** Whether the next recipe variant can be tried, and if not, why. */
4482
- recipe_exploration: PipelineRecipeExploration;
4483
3830
  /** The rule's monthly limit. Null means it has none. */
4484
3831
  monthly_ceiling_cents: number | null;
4485
3832
  /**
@@ -4496,24 +3843,10 @@ export interface Pipeline {
4496
3843
  */
4497
3844
  month_spent_cents: number;
4498
3845
  active_run: PipelineActiveRun | null;
3846
+ /** Winners only, v1 first. */
4499
3847
  versions: PipelineVersion[];
4500
- }
4501
- export interface TrainingRuleListResponse {
4502
- rules: TrainingRule[];
4503
- total: number;
4504
- }
4505
- export interface TrainingRuleResponse {
4506
- rule: TrainingRule;
4507
- recent_runs: TrainingRunSummary[];
4508
- /** What this rule's runs have cost at most this UTC calendar month; see {@link Pipeline.month_spent_cents}. */
4509
- month_spent_cents: number;
4510
- judge_agreement: JudgeAgreement | null;
4511
- }
4512
- export interface TrainingRuleMutationResponse {
4513
- rule: TrainingRule;
4514
- /** The rule will not fire again until someone confirms the new amounts. */
4515
- consent_required: boolean;
4516
- preflight: TrainingRulePreflight | null;
3848
+ /** Every attempt, attempt 1 first. */
3849
+ attempts: PipelineAttempt[];
4517
3850
  }
4518
3851
  export interface TrainingRuleDeleteResponse {
4519
3852
  deleted: boolean;
@@ -4552,9 +3885,6 @@ export interface EvaluationItemsResponse {
4552
3885
  items: EvaluationItem[];
4553
3886
  total: number;
4554
3887
  }
4555
- export interface AgentSettingsResponse {
4556
- settings: AgentSettings;
4557
- }
4558
3888
  export interface InferenceAliasListResponse {
4559
3889
  aliases: InferenceAlias[];
4560
3890
  }
@@ -4578,14 +3908,234 @@ export interface BenchmarkHistoryResponse {
4578
3908
  points: BenchmarkHistoryPoint[];
4579
3909
  total: number;
4580
3910
  }
3911
+ export interface PipelineListParams {
3912
+ /** How many of the newest pipelines to skip: the page after one that started at N starts at N + its length. */
3913
+ offset?: number;
3914
+ }
4581
3915
  export interface PipelineListResponse {
4582
3916
  /** The newest pipelines first, at most the service's page (100). */
4583
3917
  pipelines: Pipeline[];
4584
3918
  /** Every pipeline in the workspace, including any not returned. */
4585
3919
  total: number;
4586
- /** More pipelines exist than were returned; the rest are reached through `listTrainingRules`. */
3920
+ /** More pipelines exist after this page. */
4587
3921
  truncated: boolean;
4588
3922
  }
4589
3923
  export interface PipelineResponse {
4590
3924
  pipeline: Pipeline;
4591
3925
  }
3926
+ /**
3927
+ * The body of `POST /api/loop/pipelines` and (every field optional)
3928
+ * `PUT /api/loop/pipelines/{id}`. Everything else -- the machines, the money,
3929
+ * the adapter, the held-back set, the consent -- the platform decides; a field
3930
+ * not listed here is refused by name. See the loop-service README, "Simple
3931
+ * pipelines".
3932
+ */
3933
+ export interface PipelineWriteRequest {
3934
+ workspace_id?: string;
3935
+ /** Required on a create. The pipeline's own tag is this name made into a tag. */
3936
+ name?: string;
3937
+ use_case?: string;
3938
+ /** A model from {@link LoopModel}, by id. The server pins its revision. */
3939
+ model?: {
3940
+ id: string;
3941
+ };
3942
+ /** A second model to train on the same data; null removes it. */
3943
+ challenger_model?: {
3944
+ id: string;
3945
+ } | null;
3946
+ /** More tags whose conversations the pipeline trains on. On an edit the list replaces the old one. */
3947
+ tags?: string[];
3948
+ /** At least `min_samples` new samples (250 and up for SFT) AND, when set, the schedule. */
3949
+ train_when?: {
3950
+ min_samples?: number;
3951
+ schedule?: {
3952
+ every: 'day' | 'week';
3953
+ weekday?: number;
3954
+ hour_utc: number;
3955
+ } | null;
3956
+ };
3957
+ promotion?: {
3958
+ policy: PipelinePromotion['policy'];
3959
+ k?: number;
3960
+ };
3961
+ /** 1 to 100, or null for no limit (the default). Counts versions, not attempts. */
3962
+ max_versions?: number | null;
3963
+ enabled?: boolean;
3964
+ /** An edit's compare-and-set. */
3965
+ expected_revision?: number;
3966
+ }
3967
+ /** What a pipeline create, edit or train-now answers. */
3968
+ export interface PipelineMutationResponse {
3969
+ pipeline: Pipeline;
3970
+ }
3971
+ /** One base model a pipeline can train (`GET /api/loop/models`). */
3972
+ export interface LoopModel {
3973
+ /** The repository id a pipeline body names in `model.id`. */
3974
+ id: string;
3975
+ name: string;
3976
+ author: string;
3977
+ family: string;
3978
+ /** Total parameters, in billions. */
3979
+ params_b: number;
3980
+ /** Verified end to end on the platform for training and serving. */
3981
+ recommended: boolean;
3982
+ }
3983
+ /** The floors a new pipeline is held to, in samples (`GET /api/loop/models`). */
3984
+ export interface PipelineFloors {
3985
+ /** Samples a new pipeline's first attempt needs: training rows plus the held-back set. */
3986
+ first_version_samples: number;
3987
+ /** `train_when.min_samples`' default and floor. */
3988
+ new_samples: number;
3989
+ }
3990
+ export interface LoopModelsResponse {
3991
+ models: LoopModel[];
3992
+ floors: PipelineFloors;
3993
+ }
3994
+ /** The promotion rule as the API reads it back. */
3995
+ export interface PromotionPolicyState {
3996
+ policy: string;
3997
+ /** Under k_of_n, how many measurements must improve; null otherwise. */
3998
+ k: number | null;
3999
+ updated_at: string | null;
4000
+ updated_by: string | null;
4001
+ }
4002
+ /** One benchmark on a pipeline's list, attached now or once. */
4003
+ export interface RuleBenchmark {
4004
+ benchmark_id: string;
4005
+ name: string;
4006
+ benchmark_status: string;
4007
+ item_count: number;
4008
+ items_digest: string;
4009
+ /** False for a benchmark this pipeline stopped using; its history stays. */
4010
+ attached: boolean;
4011
+ /** 1 is the primary; null once removed. */
4012
+ priority: number | null;
4013
+ primary: boolean;
4014
+ added_at: string;
4015
+ added_by: string | null;
4016
+ removed_at: string | null;
4017
+ removed_by: string | null;
4018
+ }
4019
+ /** What every `/api/loop/pipelines/{id}/benchmarks` route answers. */
4020
+ export interface PipelineBenchmarksResponse {
4021
+ rule_id: string;
4022
+ promotion_policy: PromotionPolicyState;
4023
+ /** Attached ones in priority order, then removed ones, newest first. */
4024
+ benchmarks: RuleBenchmark[];
4025
+ }
4026
+ /** `POST /api/loop/pipelines/{id}/benchmarks`. */
4027
+ export interface PipelineBenchmarkAttachParams {
4028
+ benchmark_id: string;
4029
+ /** Where it goes on the list, 1 first. Absent puts it last. */
4030
+ priority?: number;
4031
+ }
4032
+ /** `PUT /api/loop/pipelines/{id}/benchmarks/{benchmark_id}`. */
4033
+ export interface PipelineBenchmarkMoveParams {
4034
+ /** 1 makes it the primary. */
4035
+ priority: number;
4036
+ }
4037
+ /** A benchmark any version of this pipeline was, or is now, measured on. */
4038
+ export interface CompareBenchmark {
4039
+ benchmark_id: string;
4040
+ name: string;
4041
+ benchmark_status: string;
4042
+ items_digest: string;
4043
+ attached: boolean;
4044
+ priority: number | null;
4045
+ added_at: string | null;
4046
+ removed_at: string | null;
4047
+ }
4048
+ /** One version on one benchmark. `why` says why there is no score. */
4049
+ export interface VersionBenchmarkScore {
4050
+ benchmark_id: string;
4051
+ /** `scored`, `no_score`, or `not_measured` (the benchmark was added after this version). */
4052
+ state: string;
4053
+ benchmark_run_id: string | null;
4054
+ replay_status: string | null;
4055
+ candidate_score: number | null;
4056
+ incumbent_score: number | null;
4057
+ score_delta: number | null;
4058
+ rows_scored: number | null;
4059
+ rows_total: number | null;
4060
+ items_digest: string | null;
4061
+ comparable: boolean;
4062
+ why: string | null;
4063
+ }
4064
+ /** One measurement a promotion rule weighed. */
4065
+ export interface PromotionMeasurement {
4066
+ kind: string;
4067
+ benchmark_id: string | null;
4068
+ benchmark_name: string | null;
4069
+ benchmark_run_id: string | null;
4070
+ items_digest: string | null;
4071
+ priority: number | null;
4072
+ weight: number;
4073
+ delta: number | null;
4074
+ outcome: string;
4075
+ reason: string;
4076
+ }
4077
+ /** What a version's promotion rule decided, and on which numbers. */
4078
+ export interface PromotionOutcome {
4079
+ policy: string;
4080
+ k: number | null;
4081
+ verdict: string;
4082
+ decided_by: string;
4083
+ reason: string;
4084
+ auto_promote_allowed: boolean;
4085
+ weighted_delta: number | null;
4086
+ measurements: PromotionMeasurement[];
4087
+ }
4088
+ /** One version with every measurement it has. */
4089
+ export interface VersionScores {
4090
+ version: number;
4091
+ run_id: string;
4092
+ /** The attempt that became this version, and the base model it trained. */
4093
+ attempt: number | null;
4094
+ base_model_id: string | null;
4095
+ model_family: string | null;
4096
+ state: string;
4097
+ verdict: string | null;
4098
+ decision: string | null;
4099
+ compared_at: string | null;
4100
+ holdout: {
4101
+ rows_scored: number | null;
4102
+ win_rate: number | null;
4103
+ mean_delta: number | null;
4104
+ };
4105
+ /** One entry per entry of the response's `benchmarks`, in the same order. */
4106
+ benchmarks: VersionBenchmarkScore[];
4107
+ promotion: PromotionOutcome | null;
4108
+ }
4109
+ /** Version b against version a on one benchmark. `delta` is b minus a. */
4110
+ export interface VersionScoreDifference {
4111
+ benchmark_id: string;
4112
+ a_score: number | null;
4113
+ b_score: number | null;
4114
+ delta: number | null;
4115
+ comparable: boolean;
4116
+ why: string | null;
4117
+ }
4118
+ /** `GET /api/loop/pipelines/{id}/versions/compare`. */
4119
+ export interface VersionCompareResponse {
4120
+ rule_id: string;
4121
+ promotion_policy: PromotionPolicyState;
4122
+ benchmarks: CompareBenchmark[];
4123
+ versions: VersionScores[];
4124
+ a: number | null;
4125
+ b: number | null;
4126
+ differences: VersionScoreDifference[];
4127
+ }
4128
+ /** Which two versions to set side by side, by VERSION number (v1 is 1). */
4129
+ export interface VersionCompareParams {
4130
+ a?: number;
4131
+ b?: number;
4132
+ }
4133
+ /**
4134
+ * `POST /api/loop/pipelines/{id}/import`: the import body, `source` optional.
4135
+ * `default_feedback: 'good'` records every answered row without a verdict of
4136
+ * its own as a good example to learn from.
4137
+ */
4138
+ export type PipelineImportParams = Omit<LoopImportParams, 'source'> & {
4139
+ source?: string;
4140
+ default_feedback?: 'good';
4141
+ };