runbios-sdk 0.2.1-dev.112 → 0.2.1-dev.114
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/dist/index.d.ts +1 -1
- package/dist/index.js +1 -1
- package/dist/resources/loop.d.ts +16 -3
- package/dist/resources/loop.js +31 -5
- package/dist/types.d.ts +21 -1
- package/package.json +1 -1
package/dist/index.d.ts
CHANGED
|
@@ -36,7 +36,7 @@ import { Loop } from './resources/loop.js';
|
|
|
36
36
|
* SDK version. Sent as part of the User-Agent header.
|
|
37
37
|
* Must match package.json "version" -- enforced by a contract test.
|
|
38
38
|
*/
|
|
39
|
-
export declare const VERSION = "0.2.1-dev.
|
|
39
|
+
export declare const VERSION = "0.2.1-dev.114";
|
|
40
40
|
export declare class RunBiOS {
|
|
41
41
|
/** Search models, fetch configs, check adapter compatibility. */
|
|
42
42
|
readonly models: Models;
|
package/dist/index.js
CHANGED
|
@@ -36,7 +36,7 @@ import { Loop } from './resources/loop.js';
|
|
|
36
36
|
* SDK version. Sent as part of the User-Agent header.
|
|
37
37
|
* Must match package.json "version" -- enforced by a contract test.
|
|
38
38
|
*/
|
|
39
|
-
export const VERSION = '0.2.1-dev.
|
|
39
|
+
export const VERSION = '0.2.1-dev.114';
|
|
40
40
|
export class RunBiOS {
|
|
41
41
|
/** Search models, fetch configs, check adapter compatibility. */
|
|
42
42
|
models;
|
package/dist/resources/loop.d.ts
CHANGED
|
@@ -171,9 +171,18 @@ export declare class Loop {
|
|
|
171
171
|
*
|
|
172
172
|
* Labels are lowercased and trimmed, so `Refunds` and `refunds` are one label.
|
|
173
173
|
*/
|
|
174
|
-
label(traceId: string, labels: string[], parent?: string): Promise<LoopLabel[]>;
|
|
175
|
-
|
|
174
|
+
label(traceId: string, labels: string[], parent?: string, attributes?: Record<string, string>): Promise<LoopLabel[]>;
|
|
175
|
+
/**
|
|
176
|
+
* Set named dimensions on a conversation: `{ category: 'billing',
|
|
177
|
+
* language: 'es' }`.
|
|
178
|
+
*
|
|
179
|
+
* An upsert per dimension, so correcting `category` leaves `task` and any
|
|
180
|
+
* bare tags exactly where they were — no delete-then-add.
|
|
181
|
+
*/
|
|
182
|
+
setAttributes(traceId: string, attributes: Record<string, string>): Promise<LoopLabel[]>;
|
|
183
|
+
unlabel(traceId: string, label: string, key?: string): Promise<{
|
|
176
184
|
deleted: boolean;
|
|
185
|
+
key: string;
|
|
177
186
|
label: string;
|
|
178
187
|
}>;
|
|
179
188
|
/** Every label in the workspace, with how many conversations carry it. */
|
|
@@ -187,7 +196,11 @@ export declare class Loop {
|
|
|
187
196
|
* - `sft` -- answers you marked right, and answers you rewrote.
|
|
188
197
|
* - `dpo` -- answers you REWROTE, so a better and a worse version of the same
|
|
189
198
|
* reply exist. Nothing else produces a pair.
|
|
190
|
-
* - `grpo` -- answers with a value or fact they can be checked against
|
|
199
|
+
* - `grpo` -- answers with a value or fact they can be checked against, or
|
|
200
|
+
* a judge rubric, which scores answers the run has not written yet.
|
|
201
|
+
* - `kto` -- anything carrying a yes or a no, INCLUDING a thumbs-down with
|
|
202
|
+
* nothing written. That row trains nothing under the other three methods,
|
|
203
|
+
* which is why this one exists: it is the feedback people actually give.
|
|
191
204
|
*
|
|
192
205
|
* `holdout_percent` holds back your most recent work rather than a random
|
|
193
206
|
* slice, so the evaluation measures whether the model generalised instead of
|
package/dist/resources/loop.js
CHANGED
|
@@ -76,6 +76,16 @@ export class Loop {
|
|
|
76
76
|
q.set('to', params.to);
|
|
77
77
|
if (params.signalled)
|
|
78
78
|
q.set('signalled', 'true');
|
|
79
|
+
if (params.label)
|
|
80
|
+
q.set('label', params.label);
|
|
81
|
+
// Repeated rather than comma-joined: a value containing a comma would
|
|
82
|
+
// otherwise silently split into two filters that match nothing.
|
|
83
|
+
for (const [k, v] of Object.entries(params.attributes ?? {})) {
|
|
84
|
+
if (k && v)
|
|
85
|
+
q.append('attr', `${k}:${v}`);
|
|
86
|
+
}
|
|
87
|
+
if (params.unlabelled)
|
|
88
|
+
q.set('unlabelled', 'true');
|
|
79
89
|
if (params.limit != null)
|
|
80
90
|
q.set('limit', String(params.limit));
|
|
81
91
|
if (params.offset != null)
|
|
@@ -218,12 +228,24 @@ export class Loop {
|
|
|
218
228
|
*
|
|
219
229
|
* Labels are lowercased and trimmed, so `Refunds` and `refunds` are one label.
|
|
220
230
|
*/
|
|
221
|
-
async label(traceId, labels, parent) {
|
|
222
|
-
const res = await this._http.fetchPost(`/api/loop/traces/${encodeURIComponent(traceId)}/labels`, { labels, parent });
|
|
231
|
+
async label(traceId, labels, parent, attributes) {
|
|
232
|
+
const res = await this._http.fetchPost(`/api/loop/traces/${encodeURIComponent(traceId)}/labels`, { labels, parent, attributes });
|
|
223
233
|
return res.labels;
|
|
224
234
|
}
|
|
225
|
-
|
|
226
|
-
|
|
235
|
+
/**
|
|
236
|
+
* Set named dimensions on a conversation: `{ category: 'billing',
|
|
237
|
+
* language: 'es' }`.
|
|
238
|
+
*
|
|
239
|
+
* An upsert per dimension, so correcting `category` leaves `task` and any
|
|
240
|
+
* bare tags exactly where they were — no delete-then-add.
|
|
241
|
+
*/
|
|
242
|
+
async setAttributes(traceId, attributes) {
|
|
243
|
+
const res = await this._http.fetchPost(`/api/loop/traces/${encodeURIComponent(traceId)}/labels`, { attributes });
|
|
244
|
+
return res.labels;
|
|
245
|
+
}
|
|
246
|
+
async unlabel(traceId, label, key) {
|
|
247
|
+
const q = key ? `?key=${encodeURIComponent(key)}` : '';
|
|
248
|
+
return this._http.fetchDelete(`/api/loop/traces/${encodeURIComponent(traceId)}/labels/${encodeURIComponent(label)}${q}`);
|
|
227
249
|
}
|
|
228
250
|
/** Every label in the workspace, with how many conversations carry it. */
|
|
229
251
|
async listLabels() {
|
|
@@ -240,7 +262,11 @@ export class Loop {
|
|
|
240
262
|
* - `sft` -- answers you marked right, and answers you rewrote.
|
|
241
263
|
* - `dpo` -- answers you REWROTE, so a better and a worse version of the same
|
|
242
264
|
* reply exist. Nothing else produces a pair.
|
|
243
|
-
* - `grpo` -- answers with a value or fact they can be checked against
|
|
265
|
+
* - `grpo` -- answers with a value or fact they can be checked against, or
|
|
266
|
+
* a judge rubric, which scores answers the run has not written yet.
|
|
267
|
+
* - `kto` -- anything carrying a yes or a no, INCLUDING a thumbs-down with
|
|
268
|
+
* nothing written. That row trains nothing under the other three methods,
|
|
269
|
+
* which is why this one exists: it is the feedback people actually give.
|
|
244
270
|
*
|
|
245
271
|
* `holdout_percent` holds back your most recent work rather than a random
|
|
246
272
|
* slice, so the evaluation measures whether the model generalised instead of
|
package/dist/types.d.ts
CHANGED
|
@@ -1959,7 +1959,7 @@ export interface ChatCompletionUsage {
|
|
|
1959
1959
|
prompt_tokens_details?: PromptTokensDetails;
|
|
1960
1960
|
}
|
|
1961
1961
|
/** What a training set is shaped for. The three need different things. */
|
|
1962
|
-
export type LoopMethod = 'sft' | 'dpo' | 'grpo';
|
|
1962
|
+
export type LoopMethod = 'sft' | 'dpo' | 'grpo' | 'kto';
|
|
1963
1963
|
/** Who judged an answer. */
|
|
1964
1964
|
export type LoopSource = 'human' | 'verifier' | 'judge' | 'behavioural';
|
|
1965
1965
|
/** The judgement itself. */
|
|
@@ -2073,6 +2073,15 @@ export interface LoopTraceListParams {
|
|
|
2073
2073
|
to?: string;
|
|
2074
2074
|
/** Only conversations that already carry a verdict. */
|
|
2075
2075
|
signalled?: boolean;
|
|
2076
|
+
/** One bare tag. A tag with children matches them too. */
|
|
2077
|
+
label?: string;
|
|
2078
|
+
/**
|
|
2079
|
+
* Named dimensions, ANDed together: `{ category: 'billing', language: 'es' }`.
|
|
2080
|
+
* Exact within their key — `source=support` does not match `team=support`.
|
|
2081
|
+
*/
|
|
2082
|
+
attributes?: Record<string, string>;
|
|
2083
|
+
/** Only the conversations nobody has described yet. */
|
|
2084
|
+
unlabelled?: boolean;
|
|
2076
2085
|
limit?: number;
|
|
2077
2086
|
offset?: number;
|
|
2078
2087
|
}
|
|
@@ -2095,6 +2104,14 @@ export interface LoopDatasetCreateParams {
|
|
|
2095
2104
|
/** Holds back your most recent work, not a random slice. 0-50. */
|
|
2096
2105
|
holdout_percent?: number;
|
|
2097
2106
|
max_items?: number;
|
|
2107
|
+
/**
|
|
2108
|
+
* Named dimensions, ANDed with each other and with `label`:
|
|
2109
|
+
* `{ category: 'billing', language: 'es' }`. This is what turns one captured
|
|
2110
|
+
* corpus into a different dataset for every task somebody trains for.
|
|
2111
|
+
*/
|
|
2112
|
+
attributes?: Record<string, string>;
|
|
2113
|
+
/** Only the conversations nobody has described yet. */
|
|
2114
|
+
unlabelled?: boolean;
|
|
2098
2115
|
}
|
|
2099
2116
|
export interface LoopDataset {
|
|
2100
2117
|
id: string;
|
|
@@ -2358,9 +2375,12 @@ export interface LoopLabel {
|
|
|
2358
2375
|
parent: string | null;
|
|
2359
2376
|
created_by: string | null;
|
|
2360
2377
|
created_at: string;
|
|
2378
|
+
/** The dimension this value belongs to. `tag` for a bare name. */
|
|
2379
|
+
key: string;
|
|
2361
2380
|
}
|
|
2362
2381
|
export interface LoopLabelCount {
|
|
2363
2382
|
label: string;
|
|
2364
2383
|
parent: string | null;
|
|
2365
2384
|
traces: number;
|
|
2385
|
+
key: string;
|
|
2366
2386
|
}
|