runbios-sdk 0.2.1-dev.112 → 0.2.1-dev.114

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/dist/index.d.ts CHANGED
@@ -36,7 +36,7 @@ import { Loop } from './resources/loop.js';
36
36
  * SDK version. Sent as part of the User-Agent header.
37
37
  * Must match package.json "version" -- enforced by a contract test.
38
38
  */
39
- export declare const VERSION = "0.2.1-dev.112";
39
+ export declare const VERSION = "0.2.1-dev.114";
40
40
  export declare class RunBiOS {
41
41
  /** Search models, fetch configs, check adapter compatibility. */
42
42
  readonly models: Models;
package/dist/index.js CHANGED
@@ -36,7 +36,7 @@ import { Loop } from './resources/loop.js';
36
36
  * SDK version. Sent as part of the User-Agent header.
37
37
  * Must match package.json "version" -- enforced by a contract test.
38
38
  */
39
- export const VERSION = '0.2.1-dev.112';
39
+ export const VERSION = '0.2.1-dev.114';
40
40
  export class RunBiOS {
41
41
  /** Search models, fetch configs, check adapter compatibility. */
42
42
  models;
@@ -171,9 +171,18 @@ export declare class Loop {
171
171
  *
172
172
  * Labels are lowercased and trimmed, so `Refunds` and `refunds` are one label.
173
173
  */
174
- label(traceId: string, labels: string[], parent?: string): Promise<LoopLabel[]>;
175
- unlabel(traceId: string, label: string): Promise<{
174
+ label(traceId: string, labels: string[], parent?: string, attributes?: Record<string, string>): Promise<LoopLabel[]>;
175
+ /**
176
+ * Set named dimensions on a conversation: `{ category: 'billing',
177
+ * language: 'es' }`.
178
+ *
179
+ * An upsert per dimension, so correcting `category` leaves `task` and any
180
+ * bare tags exactly where they were — no delete-then-add.
181
+ */
182
+ setAttributes(traceId: string, attributes: Record<string, string>): Promise<LoopLabel[]>;
183
+ unlabel(traceId: string, label: string, key?: string): Promise<{
176
184
  deleted: boolean;
185
+ key: string;
177
186
  label: string;
178
187
  }>;
179
188
  /** Every label in the workspace, with how many conversations carry it. */
@@ -187,7 +196,11 @@ export declare class Loop {
187
196
  * - `sft` -- answers you marked right, and answers you rewrote.
188
197
  * - `dpo` -- answers you REWROTE, so a better and a worse version of the same
189
198
  * reply exist. Nothing else produces a pair.
190
- * - `grpo` -- answers with a value or fact they can be checked against.
199
+ * - `grpo` -- answers with a value or fact they can be checked against, or
200
+ * a judge rubric, which scores answers the run has not written yet.
201
+ * - `kto` -- anything carrying a yes or a no, INCLUDING a thumbs-down with
202
+ * nothing written. That row trains nothing under the other three methods,
203
+ * which is why this one exists: it is the feedback people actually give.
191
204
  *
192
205
  * `holdout_percent` holds back your most recent work rather than a random
193
206
  * slice, so the evaluation measures whether the model generalised instead of
@@ -76,6 +76,16 @@ export class Loop {
76
76
  q.set('to', params.to);
77
77
  if (params.signalled)
78
78
  q.set('signalled', 'true');
79
+ if (params.label)
80
+ q.set('label', params.label);
81
+ // Repeated rather than comma-joined: a value containing a comma would
82
+ // otherwise silently split into two filters that match nothing.
83
+ for (const [k, v] of Object.entries(params.attributes ?? {})) {
84
+ if (k && v)
85
+ q.append('attr', `${k}:${v}`);
86
+ }
87
+ if (params.unlabelled)
88
+ q.set('unlabelled', 'true');
79
89
  if (params.limit != null)
80
90
  q.set('limit', String(params.limit));
81
91
  if (params.offset != null)
@@ -218,12 +228,24 @@ export class Loop {
218
228
  *
219
229
  * Labels are lowercased and trimmed, so `Refunds` and `refunds` are one label.
220
230
  */
221
- async label(traceId, labels, parent) {
222
- const res = await this._http.fetchPost(`/api/loop/traces/${encodeURIComponent(traceId)}/labels`, { labels, parent });
231
+ async label(traceId, labels, parent, attributes) {
232
+ const res = await this._http.fetchPost(`/api/loop/traces/${encodeURIComponent(traceId)}/labels`, { labels, parent, attributes });
223
233
  return res.labels;
224
234
  }
225
- async unlabel(traceId, label) {
226
- return this._http.fetchDelete(`/api/loop/traces/${encodeURIComponent(traceId)}/labels/${encodeURIComponent(label)}`);
235
+ /**
236
+ * Set named dimensions on a conversation: `{ category: 'billing',
237
+ * language: 'es' }`.
238
+ *
239
+ * An upsert per dimension, so correcting `category` leaves `task` and any
240
+ * bare tags exactly where they were — no delete-then-add.
241
+ */
242
+ async setAttributes(traceId, attributes) {
243
+ const res = await this._http.fetchPost(`/api/loop/traces/${encodeURIComponent(traceId)}/labels`, { attributes });
244
+ return res.labels;
245
+ }
246
+ async unlabel(traceId, label, key) {
247
+ const q = key ? `?key=${encodeURIComponent(key)}` : '';
248
+ return this._http.fetchDelete(`/api/loop/traces/${encodeURIComponent(traceId)}/labels/${encodeURIComponent(label)}${q}`);
227
249
  }
228
250
  /** Every label in the workspace, with how many conversations carry it. */
229
251
  async listLabels() {
@@ -240,7 +262,11 @@ export class Loop {
240
262
  * - `sft` -- answers you marked right, and answers you rewrote.
241
263
  * - `dpo` -- answers you REWROTE, so a better and a worse version of the same
242
264
  * reply exist. Nothing else produces a pair.
243
- * - `grpo` -- answers with a value or fact they can be checked against.
265
+ * - `grpo` -- answers with a value or fact they can be checked against, or
266
+ * a judge rubric, which scores answers the run has not written yet.
267
+ * - `kto` -- anything carrying a yes or a no, INCLUDING a thumbs-down with
268
+ * nothing written. That row trains nothing under the other three methods,
269
+ * which is why this one exists: it is the feedback people actually give.
244
270
  *
245
271
  * `holdout_percent` holds back your most recent work rather than a random
246
272
  * slice, so the evaluation measures whether the model generalised instead of
package/dist/types.d.ts CHANGED
@@ -1959,7 +1959,7 @@ export interface ChatCompletionUsage {
1959
1959
  prompt_tokens_details?: PromptTokensDetails;
1960
1960
  }
1961
1961
  /** What a training set is shaped for. The three need different things. */
1962
- export type LoopMethod = 'sft' | 'dpo' | 'grpo';
1962
+ export type LoopMethod = 'sft' | 'dpo' | 'grpo' | 'kto';
1963
1963
  /** Who judged an answer. */
1964
1964
  export type LoopSource = 'human' | 'verifier' | 'judge' | 'behavioural';
1965
1965
  /** The judgement itself. */
@@ -2073,6 +2073,15 @@ export interface LoopTraceListParams {
2073
2073
  to?: string;
2074
2074
  /** Only conversations that already carry a verdict. */
2075
2075
  signalled?: boolean;
2076
+ /** One bare tag. A tag with children matches them too. */
2077
+ label?: string;
2078
+ /**
2079
+ * Named dimensions, ANDed together: `{ category: 'billing', language: 'es' }`.
2080
+ * Exact within their key — `source=support` does not match `team=support`.
2081
+ */
2082
+ attributes?: Record<string, string>;
2083
+ /** Only the conversations nobody has described yet. */
2084
+ unlabelled?: boolean;
2076
2085
  limit?: number;
2077
2086
  offset?: number;
2078
2087
  }
@@ -2095,6 +2104,14 @@ export interface LoopDatasetCreateParams {
2095
2104
  /** Holds back your most recent work, not a random slice. 0-50. */
2096
2105
  holdout_percent?: number;
2097
2106
  max_items?: number;
2107
+ /**
2108
+ * Named dimensions, ANDed with each other and with `label`:
2109
+ * `{ category: 'billing', language: 'es' }`. This is what turns one captured
2110
+ * corpus into a different dataset for every task somebody trains for.
2111
+ */
2112
+ attributes?: Record<string, string>;
2113
+ /** Only the conversations nobody has described yet. */
2114
+ unlabelled?: boolean;
2098
2115
  }
2099
2116
  export interface LoopDataset {
2100
2117
  id: string;
@@ -2358,9 +2375,12 @@ export interface LoopLabel {
2358
2375
  parent: string | null;
2359
2376
  created_by: string | null;
2360
2377
  created_at: string;
2378
+ /** The dimension this value belongs to. `tag` for a bare name. */
2379
+ key: string;
2361
2380
  }
2362
2381
  export interface LoopLabelCount {
2363
2382
  label: string;
2364
2383
  parent: string | null;
2365
2384
  traces: number;
2385
+ key: string;
2366
2386
  }
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "runbios-sdk",
3
- "version": "0.2.1-dev.112",
3
+ "version": "0.2.1-dev.114",
4
4
  "description": "Official TypeScript SDK for the Run BiOS training and deployment platform API",
5
5
  "type": "module",
6
6
  "main": "./dist/index.js",