runbios-sdk 0.2.1-dev.112 → 0.2.1-dev.113

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/dist/index.d.ts CHANGED
@@ -36,7 +36,7 @@ import { Loop } from './resources/loop.js';
36
36
  * SDK version. Sent as part of the User-Agent header.
37
37
  * Must match package.json "version" -- enforced by a contract test.
38
38
  */
39
- export declare const VERSION = "0.2.1-dev.112";
39
+ export declare const VERSION = "0.2.1-dev.113";
40
40
  export declare class RunBiOS {
41
41
  /** Search models, fetch configs, check adapter compatibility. */
42
42
  readonly models: Models;
package/dist/index.js CHANGED
@@ -36,7 +36,7 @@ import { Loop } from './resources/loop.js';
36
36
  * SDK version. Sent as part of the User-Agent header.
37
37
  * Must match package.json "version" -- enforced by a contract test.
38
38
  */
39
- export const VERSION = '0.2.1-dev.112';
39
+ export const VERSION = '0.2.1-dev.113';
40
40
  export class RunBiOS {
41
41
  /** Search models, fetch configs, check adapter compatibility. */
42
42
  models;
@@ -187,7 +187,11 @@ export declare class Loop {
187
187
  * - `sft` -- answers you marked right, and answers you rewrote.
188
188
  * - `dpo` -- answers you REWROTE, so a better and a worse version of the same
189
189
  * reply exist. Nothing else produces a pair.
190
- * - `grpo` -- answers with a value or fact they can be checked against.
190
+ * - `grpo` -- answers with a value or fact they can be checked against, or
191
+ * a judge rubric, which scores answers the run has not written yet.
192
+ * - `kto` -- anything carrying a yes or a no, INCLUDING a thumbs-down with
193
+ * nothing written. That row trains nothing under the other three methods,
194
+ * which is why this one exists: it is the feedback people actually give.
191
195
  *
192
196
  * `holdout_percent` holds back your most recent work rather than a random
193
197
  * slice, so the evaluation measures whether the model generalised instead of
@@ -240,7 +240,11 @@ export class Loop {
240
240
  * - `sft` -- answers you marked right, and answers you rewrote.
241
241
  * - `dpo` -- answers you REWROTE, so a better and a worse version of the same
242
242
  * reply exist. Nothing else produces a pair.
243
- * - `grpo` -- answers with a value or fact they can be checked against.
243
+ * - `grpo` -- answers with a value or fact they can be checked against, or
244
+ * a judge rubric, which scores answers the run has not written yet.
245
+ * - `kto` -- anything carrying a yes or a no, INCLUDING a thumbs-down with
246
+ * nothing written. That row trains nothing under the other three methods,
247
+ * which is why this one exists: it is the feedback people actually give.
244
248
  *
245
249
  * `holdout_percent` holds back your most recent work rather than a random
246
250
  * slice, so the evaluation measures whether the model generalised instead of
package/dist/types.d.ts CHANGED
@@ -1959,7 +1959,7 @@ export interface ChatCompletionUsage {
1959
1959
  prompt_tokens_details?: PromptTokensDetails;
1960
1960
  }
1961
1961
  /** What a training set is shaped for. The three need different things. */
1962
- export type LoopMethod = 'sft' | 'dpo' | 'grpo';
1962
+ export type LoopMethod = 'sft' | 'dpo' | 'grpo' | 'kto';
1963
1963
  /** Who judged an answer. */
1964
1964
  export type LoopSource = 'human' | 'verifier' | 'judge' | 'behavioural';
1965
1965
  /** The judgement itself. */
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "runbios-sdk",
3
- "version": "0.2.1-dev.112",
3
+ "version": "0.2.1-dev.113",
4
4
  "description": "Official TypeScript SDK for the Run BiOS training and deployment platform API",
5
5
  "type": "module",
6
6
  "main": "./dist/index.js",