runbios-sdk 0.2.1-dev.108 → 0.2.1-dev.110
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/dist/index.d.ts +1 -1
- package/dist/index.js +1 -1
- package/dist/resources/loop.d.ts +62 -1
- package/dist/resources/loop.js +105 -0
- package/dist/types.d.ts +118 -0
- package/package.json +1 -1
package/dist/index.d.ts
CHANGED
|
@@ -35,7 +35,7 @@ import { Loop } from './resources/loop.js';
|
|
|
35
35
|
* SDK version. Sent as part of the User-Agent header.
|
|
36
36
|
* Must match package.json "version" -- enforced by a contract test.
|
|
37
37
|
*/
|
|
38
|
-
export declare const VERSION = "0.2.1-dev.
|
|
38
|
+
export declare const VERSION = "0.2.1-dev.110";
|
|
39
39
|
export declare class RunBiOS {
|
|
40
40
|
/** Search models, fetch configs, check adapter compatibility. */
|
|
41
41
|
readonly models: Models;
|
package/dist/index.js
CHANGED
|
@@ -35,7 +35,7 @@ import { Loop } from './resources/loop.js';
|
|
|
35
35
|
* SDK version. Sent as part of the User-Agent header.
|
|
36
36
|
* Must match package.json "version" -- enforced by a contract test.
|
|
37
37
|
*/
|
|
38
|
-
export const VERSION = '0.2.1-dev.
|
|
38
|
+
export const VERSION = '0.2.1-dev.110';
|
|
39
39
|
export class RunBiOS {
|
|
40
40
|
/** Search models, fetch configs, check adapter compatibility. */
|
|
41
41
|
models;
|
package/dist/resources/loop.d.ts
CHANGED
|
@@ -1,5 +1,5 @@
|
|
|
1
1
|
import type { HttpClient } from '../client.js';
|
|
2
|
-
import type { LoopTrace, LoopTraceListResponse, LoopTraceListParams, LoopCaptureParams, LoopCaptureResult, LoopSignal, LoopSignalParams, LoopCandidate, LoopCandidateParams, LoopGrader, LoopGraderParams, LoopGradeResult, LoopDataset, LoopDatasetListParams, LoopDatasetItem, LoopDatasetCreateParams, LoopConfig, LoopConfigParams, LoopStats } from '../types.js';
|
|
2
|
+
import type { LoopTrace, LoopTraceListResponse, LoopTraceListParams, LoopCaptureParams, LoopCaptureResult, LoopSignal, LoopSignalParams, LoopCandidate, LoopCandidateParams, LoopGrader, LoopGraderParams, LoopLabel, LoopLabelCount, LoopGradeResult, LoopDataset, LoopDatasetListParams, LoopDatasetItem, LoopDatasetCreateParams, LoopConfig, LoopConfigParams, LoopStats, LoopJudge, LoopJudgeParams, LoopJudgeRun, LoopJudgeWork, LoopJudgeVerdict, LoopJudgeVerdictResult } from '../types.js';
|
|
3
3
|
/**
|
|
4
4
|
* The Conscious Loop -- capture what your model was asked and answered, record
|
|
5
5
|
* whether it was right, and turn those judgements into training data.
|
|
@@ -157,6 +157,27 @@ export declare class Loop {
|
|
|
157
157
|
addCandidate(traceId: string, params: LoopCandidateParams): Promise<LoopCandidate>;
|
|
158
158
|
/** Every alternative answer recorded for one prompt, oldest first. */
|
|
159
159
|
listCandidates(traceId: string): Promise<LoopCandidate[]>;
|
|
160
|
+
/**
|
|
161
|
+
* Label a conversation, so it can be selected later.
|
|
162
|
+
*
|
|
163
|
+
* An average over everything is the least useful thing to train on. A model
|
|
164
|
+
* weak at refunds is fixed with refund examples, and you can only select
|
|
165
|
+
* those if you said so when the conversation arrived.
|
|
166
|
+
*
|
|
167
|
+
* A label may name a `parent`, which is what makes a MASTER GROUP: labelling
|
|
168
|
+
* something `refunds` with parent `billing` makes it selectable as either,
|
|
169
|
+
* and selecting `billing` later gathers every child without you maintaining
|
|
170
|
+
* a list of them.
|
|
171
|
+
*
|
|
172
|
+
* Labels are lowercased and trimmed, so `Refunds` and `refunds` are one label.
|
|
173
|
+
*/
|
|
174
|
+
label(traceId: string, labels: string[], parent?: string): Promise<LoopLabel[]>;
|
|
175
|
+
unlabel(traceId: string, label: string): Promise<{
|
|
176
|
+
deleted: boolean;
|
|
177
|
+
label: string;
|
|
178
|
+
}>;
|
|
179
|
+
/** Every label in the workspace, with how many conversations carry it. */
|
|
180
|
+
listLabels(): Promise<LoopLabelCount[]>;
|
|
160
181
|
/**
|
|
161
182
|
* Build a training set from the feedback recorded so far.
|
|
162
183
|
*
|
|
@@ -274,4 +295,44 @@ export declare class Loop {
|
|
|
274
295
|
* one row while a set is built, so the finished count can be lower.
|
|
275
296
|
*/
|
|
276
297
|
stats(): Promise<LoopStats>;
|
|
298
|
+
/** Write a rubric: what makes an answer good here, and what to score. */
|
|
299
|
+
createJudge(params: LoopJudgeParams): Promise<LoopJudge>;
|
|
300
|
+
listJudges(): Promise<LoopJudge[]>;
|
|
301
|
+
/**
|
|
302
|
+
* Rewrite a rubric. Runs already recorded keep the instructions they used:
|
|
303
|
+
* a verdict has to keep meaning what it meant when it was given.
|
|
304
|
+
*/
|
|
305
|
+
updateJudge(judgeId: string, params: LoopJudgeParams): Promise<LoopJudge>;
|
|
306
|
+
/**
|
|
307
|
+
* Retire a rubric. Its runs go with it; the verdicts they wrote stay. A score
|
|
308
|
+
* is evidence about an answer, and it does not stop being true because the
|
|
309
|
+
* rubric was retired.
|
|
310
|
+
*/
|
|
311
|
+
deleteJudge(judgeId: string): Promise<void>;
|
|
312
|
+
/**
|
|
313
|
+
* Open a run: select the conversations now, and freeze the rubric onto them.
|
|
314
|
+
*
|
|
315
|
+
* The selection is fixed at this moment on purpose. A run whose selection is
|
|
316
|
+
* a live query silently grows as conversations arrive, so "we scored the
|
|
317
|
+
* refunds slice" becomes a claim about a set that no longer exists.
|
|
318
|
+
*/
|
|
319
|
+
startRun(judgeId: string, limit?: number): Promise<LoopJudgeRun>;
|
|
320
|
+
listRuns(judgeId?: string): Promise<LoopJudgeRun[]>;
|
|
321
|
+
getRun(runId: string): Promise<LoopJudgeRun>;
|
|
322
|
+
/**
|
|
323
|
+
* Take the next conversations to score, each rendered into a ready-to-send
|
|
324
|
+
* prompt. Nothing is marked taken, so a caller that dies half way loses
|
|
325
|
+
* nothing: ask again and the same items come back.
|
|
326
|
+
*/
|
|
327
|
+
takeWork(runId: string, limit?: number): Promise<LoopJudgeWork>;
|
|
328
|
+
/**
|
|
329
|
+
* Hand the scores back. A verdict for a conversation outside this run, or
|
|
330
|
+
* scoring something the rubric never asked for, is refused and reported in
|
|
331
|
+
* `rejected` rather than silently dropped.
|
|
332
|
+
*
|
|
333
|
+
* Pass `finish` to close a run you have decided to stop early. Without it an
|
|
334
|
+
* unfinished run stays open, which is the honest state for work that was
|
|
335
|
+
* abandoned rather than completed.
|
|
336
|
+
*/
|
|
337
|
+
postVerdicts(runId: string, verdicts: LoopJudgeVerdict[], finish?: boolean): Promise<LoopJudgeVerdictResult>;
|
|
277
338
|
}
|
package/dist/resources/loop.js
CHANGED
|
@@ -203,6 +203,33 @@ export class Loop {
|
|
|
203
203
|
const res = await this._http.fetchGet(`/api/loop/traces/${encodeURIComponent(traceId)}/candidates`);
|
|
204
204
|
return res.candidates;
|
|
205
205
|
}
|
|
206
|
+
// ── labels ────────────────────────────────────────────────────────────
|
|
207
|
+
/**
|
|
208
|
+
* Label a conversation, so it can be selected later.
|
|
209
|
+
*
|
|
210
|
+
* An average over everything is the least useful thing to train on. A model
|
|
211
|
+
* weak at refunds is fixed with refund examples, and you can only select
|
|
212
|
+
* those if you said so when the conversation arrived.
|
|
213
|
+
*
|
|
214
|
+
* A label may name a `parent`, which is what makes a MASTER GROUP: labelling
|
|
215
|
+
* something `refunds` with parent `billing` makes it selectable as either,
|
|
216
|
+
* and selecting `billing` later gathers every child without you maintaining
|
|
217
|
+
* a list of them.
|
|
218
|
+
*
|
|
219
|
+
* Labels are lowercased and trimmed, so `Refunds` and `refunds` are one label.
|
|
220
|
+
*/
|
|
221
|
+
async label(traceId, labels, parent) {
|
|
222
|
+
const res = await this._http.fetchPost(`/api/loop/traces/${encodeURIComponent(traceId)}/labels`, { labels, parent });
|
|
223
|
+
return res.labels;
|
|
224
|
+
}
|
|
225
|
+
async unlabel(traceId, label) {
|
|
226
|
+
return this._http.fetchDelete(`/api/loop/traces/${encodeURIComponent(traceId)}/labels/${encodeURIComponent(label)}`);
|
|
227
|
+
}
|
|
228
|
+
/** Every label in the workspace, with how many conversations carry it. */
|
|
229
|
+
async listLabels() {
|
|
230
|
+
const res = await this._http.fetchGet('/api/loop/labels');
|
|
231
|
+
return res.labels;
|
|
232
|
+
}
|
|
206
233
|
// ── training sets ─────────────────────────────────────────────────────
|
|
207
234
|
/**
|
|
208
235
|
* Build a training set from the feedback recorded so far.
|
|
@@ -367,4 +394,82 @@ export class Loop {
|
|
|
367
394
|
const res = await this._http.fetchGet('/api/loop/stats');
|
|
368
395
|
return res.stats;
|
|
369
396
|
}
|
|
397
|
+
// ── judges ────────────────────────────────────────────────────────────
|
|
398
|
+
//
|
|
399
|
+
// A grader is a rule: deterministic, cheap, and blind to anything it was not
|
|
400
|
+
// told to look for. A judge is a rubric handed to a model, which is the only
|
|
401
|
+
// thing that can answer "was this answer actually helpful".
|
|
402
|
+
//
|
|
403
|
+
// YOU run the model. This service stores raw prompts and completions and
|
|
404
|
+
// holds them with no outbound credentials at all, which is most of the reason
|
|
405
|
+
// it is safe to store them there -- so it hands the work out instead. Open a
|
|
406
|
+
// run, take the batch, send each `prompt` to whatever model you like, and
|
|
407
|
+
// post the scores back.
|
|
408
|
+
/** Write a rubric: what makes an answer good here, and what to score. */
|
|
409
|
+
async createJudge(params) {
|
|
410
|
+
const res = await this._http.fetchPost('/api/loop/judges', params);
|
|
411
|
+
return res.judge;
|
|
412
|
+
}
|
|
413
|
+
async listJudges() {
|
|
414
|
+
const res = await this._http.fetchGet('/api/loop/judges');
|
|
415
|
+
return res.judges;
|
|
416
|
+
}
|
|
417
|
+
/**
|
|
418
|
+
* Rewrite a rubric. Runs already recorded keep the instructions they used:
|
|
419
|
+
* a verdict has to keep meaning what it meant when it was given.
|
|
420
|
+
*/
|
|
421
|
+
async updateJudge(judgeId, params) {
|
|
422
|
+
const res = await this._http.fetchPut(`/api/loop/judges/${encodeURIComponent(judgeId)}`, params);
|
|
423
|
+
return res.judge;
|
|
424
|
+
}
|
|
425
|
+
/**
|
|
426
|
+
* Retire a rubric. Its runs go with it; the verdicts they wrote stay. A score
|
|
427
|
+
* is evidence about an answer, and it does not stop being true because the
|
|
428
|
+
* rubric was retired.
|
|
429
|
+
*/
|
|
430
|
+
async deleteJudge(judgeId) {
|
|
431
|
+
await this._http.fetchDelete(`/api/loop/judges/${encodeURIComponent(judgeId)}`);
|
|
432
|
+
}
|
|
433
|
+
/**
|
|
434
|
+
* Open a run: select the conversations now, and freeze the rubric onto them.
|
|
435
|
+
*
|
|
436
|
+
* The selection is fixed at this moment on purpose. A run whose selection is
|
|
437
|
+
* a live query silently grows as conversations arrive, so "we scored the
|
|
438
|
+
* refunds slice" becomes a claim about a set that no longer exists.
|
|
439
|
+
*/
|
|
440
|
+
async startRun(judgeId, limit) {
|
|
441
|
+
const res = await this._http.fetchPost(`/api/loop/judges/${encodeURIComponent(judgeId)}/runs`, limit ? { limit } : {});
|
|
442
|
+
return res.run;
|
|
443
|
+
}
|
|
444
|
+
async listRuns(judgeId) {
|
|
445
|
+
const path = judgeId
|
|
446
|
+
? `/api/loop/judges/${encodeURIComponent(judgeId)}/runs`
|
|
447
|
+
: '/api/loop/runs';
|
|
448
|
+
const res = await this._http.fetchGet(path);
|
|
449
|
+
return res.runs;
|
|
450
|
+
}
|
|
451
|
+
async getRun(runId) {
|
|
452
|
+
const res = await this._http.fetchGet(`/api/loop/runs/${encodeURIComponent(runId)}`);
|
|
453
|
+
return res.run;
|
|
454
|
+
}
|
|
455
|
+
/**
|
|
456
|
+
* Take the next conversations to score, each rendered into a ready-to-send
|
|
457
|
+
* prompt. Nothing is marked taken, so a caller that dies half way loses
|
|
458
|
+
* nothing: ask again and the same items come back.
|
|
459
|
+
*/
|
|
460
|
+
async takeWork(runId, limit = 20) {
|
|
461
|
+
return this._http.fetchGet(`/api/loop/runs/${encodeURIComponent(runId)}/work?limit=${limit}`);
|
|
462
|
+
}
|
|
463
|
+
/**
|
|
464
|
+
* Hand the scores back. A verdict for a conversation outside this run, or
|
|
465
|
+
* scoring something the rubric never asked for, is refused and reported in
|
|
466
|
+
* `rejected` rather than silently dropped.
|
|
467
|
+
*
|
|
468
|
+
* Pass `finish` to close a run you have decided to stop early. Without it an
|
|
469
|
+
* unfinished run stays open, which is the honest state for work that was
|
|
470
|
+
* abandoned rather than completed.
|
|
471
|
+
*/
|
|
472
|
+
async postVerdicts(runId, verdicts, finish = false) {
|
|
473
|
+
return this._http.fetchPost(`/api/loop/runs/${encodeURIComponent(runId)}/verdicts`, { verdicts, finish });
|
|
474
|
+
}
|
|
370
475
|
}
|
package/dist/types.d.ts
CHANGED
|
@@ -2085,6 +2085,10 @@ export interface LoopDatasetCreateParams {
|
|
|
2085
2085
|
deployment_id?: string;
|
|
2086
2086
|
from?: string;
|
|
2087
2087
|
to?: string;
|
|
2088
|
+
/** Narrow to one slice. A label with children selects them too. */
|
|
2089
|
+
label?: string;
|
|
2090
|
+
/** Take this many at random from what the filters matched. */
|
|
2091
|
+
sample?: number;
|
|
2088
2092
|
/** Holds back your most recent work, not a random slice. 0-50. */
|
|
2089
2093
|
holdout_percent?: number;
|
|
2090
2094
|
max_items?: number;
|
|
@@ -2138,6 +2142,108 @@ export interface LoopConfigParams {
|
|
|
2138
2142
|
/** Deterministic per conversation, so turns are never split. 0 < rate <= 1. */
|
|
2139
2143
|
sample_rate?: number;
|
|
2140
2144
|
}
|
|
2145
|
+
/** One thing a judge scores, separately from the others. */
|
|
2146
|
+
export interface LoopJudgeDimension {
|
|
2147
|
+
key: string;
|
|
2148
|
+
description?: string;
|
|
2149
|
+
}
|
|
2150
|
+
/** Which slice of the corpus a judge is responsible for. */
|
|
2151
|
+
export interface LoopJudgeSelection {
|
|
2152
|
+
label?: string;
|
|
2153
|
+
deployment_id?: string;
|
|
2154
|
+
from?: string;
|
|
2155
|
+
to?: string;
|
|
2156
|
+
sample?: number;
|
|
2157
|
+
/** Skip conversations that already carry a judge verdict. */
|
|
2158
|
+
only_unscored?: boolean;
|
|
2159
|
+
}
|
|
2160
|
+
export interface LoopJudge {
|
|
2161
|
+
id: string;
|
|
2162
|
+
workspace_id: string;
|
|
2163
|
+
name: string;
|
|
2164
|
+
instructions: string;
|
|
2165
|
+
dimensions: LoopJudgeDimension[];
|
|
2166
|
+
selection: LoopJudgeSelection;
|
|
2167
|
+
model: string | null;
|
|
2168
|
+
write_gold: boolean;
|
|
2169
|
+
enabled: boolean;
|
|
2170
|
+
created_by: string | null;
|
|
2171
|
+
created_at: string;
|
|
2172
|
+
updated_at: string;
|
|
2173
|
+
}
|
|
2174
|
+
export interface LoopJudgeParams {
|
|
2175
|
+
name: string;
|
|
2176
|
+
instructions: string;
|
|
2177
|
+
dimensions: LoopJudgeDimension[];
|
|
2178
|
+
selection?: LoopJudgeSelection;
|
|
2179
|
+
model?: string;
|
|
2180
|
+
/**
|
|
2181
|
+
* Let this judge write the answer that should have been given, not just a
|
|
2182
|
+
* score. Off by default: a reference answer written by a model and then
|
|
2183
|
+
* trained on is distillation, which is a decision you make deliberately.
|
|
2184
|
+
*/
|
|
2185
|
+
write_gold?: boolean;
|
|
2186
|
+
enabled?: boolean;
|
|
2187
|
+
}
|
|
2188
|
+
/** One pass of one rubric over one slice. */
|
|
2189
|
+
export interface LoopJudgeRun {
|
|
2190
|
+
id: string;
|
|
2191
|
+
judge_id: string;
|
|
2192
|
+
judge_name?: string;
|
|
2193
|
+
status: 'open' | 'done';
|
|
2194
|
+
instructions: string;
|
|
2195
|
+
dimensions: LoopJudgeDimension[];
|
|
2196
|
+
model: string | null;
|
|
2197
|
+
selected: number;
|
|
2198
|
+
scored: number;
|
|
2199
|
+
failed: number;
|
|
2200
|
+
created_by: string | null;
|
|
2201
|
+
created_at: string;
|
|
2202
|
+
finished_at: string | null;
|
|
2203
|
+
}
|
|
2204
|
+
/**
|
|
2205
|
+
* One conversation to score, already rendered into the prompt to send.
|
|
2206
|
+
*
|
|
2207
|
+
* Send `prompt` as-is. Assembling it yourself is how two callers end up giving
|
|
2208
|
+
* the same rubric different instructions, and two judges given different
|
|
2209
|
+
* instructions are not one judge.
|
|
2210
|
+
*/
|
|
2211
|
+
export interface LoopJudgeWorkItem {
|
|
2212
|
+
trace_id: string;
|
|
2213
|
+
messages: Array<{
|
|
2214
|
+
role: string;
|
|
2215
|
+
content?: string;
|
|
2216
|
+
}>;
|
|
2217
|
+
answer: string;
|
|
2218
|
+
prompt: string;
|
|
2219
|
+
}
|
|
2220
|
+
export interface LoopJudgeWork {
|
|
2221
|
+
run: LoopJudgeRun;
|
|
2222
|
+
items: LoopJudgeWorkItem[];
|
|
2223
|
+
remaining: number;
|
|
2224
|
+
}
|
|
2225
|
+
/** One scored conversation going back. */
|
|
2226
|
+
export interface LoopJudgeVerdict {
|
|
2227
|
+
trace_id: string;
|
|
2228
|
+
/** One score per dimension the rubric asked for, each between 0 and 1. */
|
|
2229
|
+
scores?: Record<string, number>;
|
|
2230
|
+
/** Your own combination of them. Left out, the plain mean is used. */
|
|
2231
|
+
overall?: number;
|
|
2232
|
+
reason?: string;
|
|
2233
|
+
/** Only stored if the judge was created with `write_gold`. */
|
|
2234
|
+
gold?: string;
|
|
2235
|
+
/** Mark an item you could not score, instead of dropping it silently. */
|
|
2236
|
+
error?: string;
|
|
2237
|
+
}
|
|
2238
|
+
export interface LoopJudgeVerdictResult {
|
|
2239
|
+
run: LoopJudgeRun;
|
|
2240
|
+
recorded: number;
|
|
2241
|
+
failed: number;
|
|
2242
|
+
rejected: Array<{
|
|
2243
|
+
trace_id: string;
|
|
2244
|
+
error: string;
|
|
2245
|
+
}>;
|
|
2246
|
+
}
|
|
2141
2247
|
export interface LoopStats {
|
|
2142
2248
|
traces: number;
|
|
2143
2249
|
signalled_traces: number;
|
|
@@ -2237,3 +2343,15 @@ export interface LoopGradeResult {
|
|
|
2237
2343
|
signal_id: string | null;
|
|
2238
2344
|
candidates_graded: number;
|
|
2239
2345
|
}
|
|
2346
|
+
export interface LoopLabel {
|
|
2347
|
+
label: string;
|
|
2348
|
+
/** The master group this label belongs to, if any. */
|
|
2349
|
+
parent: string | null;
|
|
2350
|
+
created_by: string | null;
|
|
2351
|
+
created_at: string;
|
|
2352
|
+
}
|
|
2353
|
+
export interface LoopLabelCount {
|
|
2354
|
+
label: string;
|
|
2355
|
+
parent: string | null;
|
|
2356
|
+
traces: number;
|
|
2357
|
+
}
|