runbios-sdk 0.2.1-dev.109 → 0.2.1-dev.111
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/dist/index.d.ts +1 -1
- package/dist/index.js +1 -1
- package/dist/resources/loop.d.ts +41 -1
- package/dist/resources/loop.js +78 -0
- package/dist/types.d.ts +108 -0
- package/package.json +1 -1
package/dist/index.d.ts
CHANGED
|
@@ -35,7 +35,7 @@ import { Loop } from './resources/loop.js';
|
|
|
35
35
|
* SDK version. Sent as part of the User-Agent header.
|
|
36
36
|
* Must match package.json "version" -- enforced by a contract test.
|
|
37
37
|
*/
|
|
38
|
-
export declare const VERSION = "0.2.1-dev.
|
|
38
|
+
export declare const VERSION = "0.2.1-dev.111";
|
|
39
39
|
export declare class RunBiOS {
|
|
40
40
|
/** Search models, fetch configs, check adapter compatibility. */
|
|
41
41
|
readonly models: Models;
|
package/dist/index.js
CHANGED
|
@@ -35,7 +35,7 @@ import { Loop } from './resources/loop.js';
|
|
|
35
35
|
* SDK version. Sent as part of the User-Agent header.
|
|
36
36
|
* Must match package.json "version" -- enforced by a contract test.
|
|
37
37
|
*/
|
|
38
|
-
export const VERSION = '0.2.1-dev.
|
|
38
|
+
export const VERSION = '0.2.1-dev.111';
|
|
39
39
|
export class RunBiOS {
|
|
40
40
|
/** Search models, fetch configs, check adapter compatibility. */
|
|
41
41
|
models;
|
package/dist/resources/loop.d.ts
CHANGED
|
@@ -1,5 +1,5 @@
|
|
|
1
1
|
import type { HttpClient } from '../client.js';
|
|
2
|
-
import type { LoopTrace, LoopTraceListResponse, LoopTraceListParams, LoopCaptureParams, LoopCaptureResult, LoopSignal, LoopSignalParams, LoopCandidate, LoopCandidateParams, LoopGrader, LoopGraderParams, LoopLabel, LoopLabelCount, LoopGradeResult, LoopDataset, LoopDatasetListParams, LoopDatasetItem, LoopDatasetCreateParams, LoopConfig, LoopConfigParams, LoopStats } from '../types.js';
|
|
2
|
+
import type { LoopTrace, LoopTraceListResponse, LoopTraceListParams, LoopCaptureParams, LoopCaptureResult, LoopSignal, LoopSignalParams, LoopCandidate, LoopCandidateParams, LoopGrader, LoopGraderParams, LoopLabel, LoopLabelCount, LoopGradeResult, LoopDataset, LoopDatasetListParams, LoopDatasetItem, LoopDatasetCreateParams, LoopConfig, LoopConfigParams, LoopStats, LoopJudge, LoopJudgeParams, LoopJudgeRun, LoopJudgeWork, LoopJudgeVerdict, LoopJudgeVerdictResult } from '../types.js';
|
|
3
3
|
/**
|
|
4
4
|
* The Conscious Loop -- capture what your model was asked and answered, record
|
|
5
5
|
* whether it was right, and turn those judgements into training data.
|
|
@@ -295,4 +295,44 @@ export declare class Loop {
|
|
|
295
295
|
* one row while a set is built, so the finished count can be lower.
|
|
296
296
|
*/
|
|
297
297
|
stats(): Promise<LoopStats>;
|
|
298
|
+
/** Write a rubric: what makes an answer good here, and what to score. */
|
|
299
|
+
createJudge(params: LoopJudgeParams): Promise<LoopJudge>;
|
|
300
|
+
listJudges(): Promise<LoopJudge[]>;
|
|
301
|
+
/**
|
|
302
|
+
* Rewrite a rubric. Runs already recorded keep the instructions they used:
|
|
303
|
+
* a verdict has to keep meaning what it meant when it was given.
|
|
304
|
+
*/
|
|
305
|
+
updateJudge(judgeId: string, params: LoopJudgeParams): Promise<LoopJudge>;
|
|
306
|
+
/**
|
|
307
|
+
* Retire a rubric. Its runs go with it; the verdicts they wrote stay. A score
|
|
308
|
+
* is evidence about an answer, and it does not stop being true because the
|
|
309
|
+
* rubric was retired.
|
|
310
|
+
*/
|
|
311
|
+
deleteJudge(judgeId: string): Promise<void>;
|
|
312
|
+
/**
|
|
313
|
+
* Open a run: select the conversations now, and freeze the rubric onto them.
|
|
314
|
+
*
|
|
315
|
+
* The selection is fixed at this moment on purpose. A run whose selection is
|
|
316
|
+
* a live query silently grows as conversations arrive, so "we scored the
|
|
317
|
+
* refunds slice" becomes a claim about a set that no longer exists.
|
|
318
|
+
*/
|
|
319
|
+
startRun(judgeId: string, limit?: number): Promise<LoopJudgeRun>;
|
|
320
|
+
listRuns(judgeId?: string): Promise<LoopJudgeRun[]>;
|
|
321
|
+
getRun(runId: string): Promise<LoopJudgeRun>;
|
|
322
|
+
/**
|
|
323
|
+
* Take the next conversations to score, each rendered into a ready-to-send
|
|
324
|
+
* prompt. Nothing is marked taken, so a caller that dies half way loses
|
|
325
|
+
* nothing: ask again and the same items come back.
|
|
326
|
+
*/
|
|
327
|
+
takeWork(runId: string, limit?: number): Promise<LoopJudgeWork>;
|
|
328
|
+
/**
|
|
329
|
+
* Hand the scores back. A verdict for a conversation outside this run, or
|
|
330
|
+
* scoring something the rubric never asked for, is refused and reported in
|
|
331
|
+
* `rejected` rather than silently dropped.
|
|
332
|
+
*
|
|
333
|
+
* Pass `finish` to close a run you have decided to stop early. Without it an
|
|
334
|
+
* unfinished run stays open, which is the honest state for work that was
|
|
335
|
+
* abandoned rather than completed.
|
|
336
|
+
*/
|
|
337
|
+
postVerdicts(runId: string, verdicts: LoopJudgeVerdict[], finish?: boolean): Promise<LoopJudgeVerdictResult>;
|
|
298
338
|
}
|
package/dist/resources/loop.js
CHANGED
|
@@ -394,4 +394,82 @@ export class Loop {
|
|
|
394
394
|
const res = await this._http.fetchGet('/api/loop/stats');
|
|
395
395
|
return res.stats;
|
|
396
396
|
}
|
|
397
|
+
// ── judges ────────────────────────────────────────────────────────────
|
|
398
|
+
//
|
|
399
|
+
// A grader is a rule: deterministic, cheap, and blind to anything it was not
|
|
400
|
+
// told to look for. A judge is a rubric handed to a model, which is the only
|
|
401
|
+
// thing that can answer "was this answer actually helpful".
|
|
402
|
+
//
|
|
403
|
+
// YOU run the model. This service stores raw prompts and completions and
|
|
404
|
+
// holds them with no outbound credentials at all, which is most of the reason
|
|
405
|
+
// it is safe to store them there -- so it hands the work out instead. Open a
|
|
406
|
+
// run, take the batch, send each `prompt` to whatever model you like, and
|
|
407
|
+
// post the scores back.
|
|
408
|
+
/** Write a rubric: what makes an answer good here, and what to score. */
|
|
409
|
+
async createJudge(params) {
|
|
410
|
+
const res = await this._http.fetchPost('/api/loop/judges', params);
|
|
411
|
+
return res.judge;
|
|
412
|
+
}
|
|
413
|
+
async listJudges() {
|
|
414
|
+
const res = await this._http.fetchGet('/api/loop/judges');
|
|
415
|
+
return res.judges;
|
|
416
|
+
}
|
|
417
|
+
/**
|
|
418
|
+
* Rewrite a rubric. Runs already recorded keep the instructions they used:
|
|
419
|
+
* a verdict has to keep meaning what it meant when it was given.
|
|
420
|
+
*/
|
|
421
|
+
async updateJudge(judgeId, params) {
|
|
422
|
+
const res = await this._http.fetchPut(`/api/loop/judges/${encodeURIComponent(judgeId)}`, params);
|
|
423
|
+
return res.judge;
|
|
424
|
+
}
|
|
425
|
+
/**
|
|
426
|
+
* Retire a rubric. Its runs go with it; the verdicts they wrote stay. A score
|
|
427
|
+
* is evidence about an answer, and it does not stop being true because the
|
|
428
|
+
* rubric was retired.
|
|
429
|
+
*/
|
|
430
|
+
async deleteJudge(judgeId) {
|
|
431
|
+
await this._http.fetchDelete(`/api/loop/judges/${encodeURIComponent(judgeId)}`);
|
|
432
|
+
}
|
|
433
|
+
/**
|
|
434
|
+
* Open a run: select the conversations now, and freeze the rubric onto them.
|
|
435
|
+
*
|
|
436
|
+
* The selection is fixed at this moment on purpose. A run whose selection is
|
|
437
|
+
* a live query silently grows as conversations arrive, so "we scored the
|
|
438
|
+
* refunds slice" becomes a claim about a set that no longer exists.
|
|
439
|
+
*/
|
|
440
|
+
async startRun(judgeId, limit) {
|
|
441
|
+
const res = await this._http.fetchPost(`/api/loop/judges/${encodeURIComponent(judgeId)}/runs`, limit ? { limit } : {});
|
|
442
|
+
return res.run;
|
|
443
|
+
}
|
|
444
|
+
async listRuns(judgeId) {
|
|
445
|
+
const path = judgeId
|
|
446
|
+
? `/api/loop/judges/${encodeURIComponent(judgeId)}/runs`
|
|
447
|
+
: '/api/loop/runs';
|
|
448
|
+
const res = await this._http.fetchGet(path);
|
|
449
|
+
return res.runs;
|
|
450
|
+
}
|
|
451
|
+
async getRun(runId) {
|
|
452
|
+
const res = await this._http.fetchGet(`/api/loop/runs/${encodeURIComponent(runId)}`);
|
|
453
|
+
return res.run;
|
|
454
|
+
}
|
|
455
|
+
/**
|
|
456
|
+
* Take the next conversations to score, each rendered into a ready-to-send
|
|
457
|
+
* prompt. Nothing is marked taken, so a caller that dies half way loses
|
|
458
|
+
* nothing: ask again and the same items come back.
|
|
459
|
+
*/
|
|
460
|
+
async takeWork(runId, limit = 20) {
|
|
461
|
+
return this._http.fetchGet(`/api/loop/runs/${encodeURIComponent(runId)}/work?limit=${limit}`);
|
|
462
|
+
}
|
|
463
|
+
/**
|
|
464
|
+
* Hand the scores back. A verdict for a conversation outside this run, or
|
|
465
|
+
* scoring something the rubric never asked for, is refused and reported in
|
|
466
|
+
* `rejected` rather than silently dropped.
|
|
467
|
+
*
|
|
468
|
+
* Pass `finish` to close a run you have decided to stop early. Without it an
|
|
469
|
+
* unfinished run stays open, which is the honest state for work that was
|
|
470
|
+
* abandoned rather than completed.
|
|
471
|
+
*/
|
|
472
|
+
async postVerdicts(runId, verdicts, finish = false) {
|
|
473
|
+
return this._http.fetchPost(`/api/loop/runs/${encodeURIComponent(runId)}/verdicts`, { verdicts, finish });
|
|
474
|
+
}
|
|
397
475
|
}
|
package/dist/types.d.ts
CHANGED
|
@@ -2142,6 +2142,108 @@ export interface LoopConfigParams {
|
|
|
2142
2142
|
/** Deterministic per conversation, so turns are never split. 0 < rate <= 1. */
|
|
2143
2143
|
sample_rate?: number;
|
|
2144
2144
|
}
|
|
2145
|
+
/** One thing a judge scores, separately from the others. */
|
|
2146
|
+
export interface LoopJudgeDimension {
|
|
2147
|
+
key: string;
|
|
2148
|
+
description?: string;
|
|
2149
|
+
}
|
|
2150
|
+
/** Which slice of the corpus a judge is responsible for. */
|
|
2151
|
+
export interface LoopJudgeSelection {
|
|
2152
|
+
label?: string;
|
|
2153
|
+
deployment_id?: string;
|
|
2154
|
+
from?: string;
|
|
2155
|
+
to?: string;
|
|
2156
|
+
sample?: number;
|
|
2157
|
+
/** Skip conversations that already carry a judge verdict. */
|
|
2158
|
+
only_unscored?: boolean;
|
|
2159
|
+
}
|
|
2160
|
+
export interface LoopJudge {
|
|
2161
|
+
id: string;
|
|
2162
|
+
workspace_id: string;
|
|
2163
|
+
name: string;
|
|
2164
|
+
instructions: string;
|
|
2165
|
+
dimensions: LoopJudgeDimension[];
|
|
2166
|
+
selection: LoopJudgeSelection;
|
|
2167
|
+
model: string | null;
|
|
2168
|
+
write_gold: boolean;
|
|
2169
|
+
enabled: boolean;
|
|
2170
|
+
created_by: string | null;
|
|
2171
|
+
created_at: string;
|
|
2172
|
+
updated_at: string;
|
|
2173
|
+
}
|
|
2174
|
+
export interface LoopJudgeParams {
|
|
2175
|
+
name: string;
|
|
2176
|
+
instructions: string;
|
|
2177
|
+
dimensions: LoopJudgeDimension[];
|
|
2178
|
+
selection?: LoopJudgeSelection;
|
|
2179
|
+
model?: string;
|
|
2180
|
+
/**
|
|
2181
|
+
* Let this judge write the answer that should have been given, not just a
|
|
2182
|
+
* score. Off by default: a reference answer written by a model and then
|
|
2183
|
+
* trained on is distillation, which is a decision you make deliberately.
|
|
2184
|
+
*/
|
|
2185
|
+
write_gold?: boolean;
|
|
2186
|
+
enabled?: boolean;
|
|
2187
|
+
}
|
|
2188
|
+
/** One pass of one rubric over one slice. */
|
|
2189
|
+
export interface LoopJudgeRun {
|
|
2190
|
+
id: string;
|
|
2191
|
+
judge_id: string;
|
|
2192
|
+
judge_name?: string;
|
|
2193
|
+
status: 'open' | 'done';
|
|
2194
|
+
instructions: string;
|
|
2195
|
+
dimensions: LoopJudgeDimension[];
|
|
2196
|
+
model: string | null;
|
|
2197
|
+
selected: number;
|
|
2198
|
+
scored: number;
|
|
2199
|
+
failed: number;
|
|
2200
|
+
created_by: string | null;
|
|
2201
|
+
created_at: string;
|
|
2202
|
+
finished_at: string | null;
|
|
2203
|
+
}
|
|
2204
|
+
/**
|
|
2205
|
+
* One conversation to score, already rendered into the prompt to send.
|
|
2206
|
+
*
|
|
2207
|
+
* Send `prompt` as-is. Assembling it yourself is how two callers end up giving
|
|
2208
|
+
* the same rubric different instructions, and two judges given different
|
|
2209
|
+
* instructions are not one judge.
|
|
2210
|
+
*/
|
|
2211
|
+
export interface LoopJudgeWorkItem {
|
|
2212
|
+
trace_id: string;
|
|
2213
|
+
messages: Array<{
|
|
2214
|
+
role: string;
|
|
2215
|
+
content?: string;
|
|
2216
|
+
}>;
|
|
2217
|
+
answer: string;
|
|
2218
|
+
prompt: string;
|
|
2219
|
+
}
|
|
2220
|
+
export interface LoopJudgeWork {
|
|
2221
|
+
run: LoopJudgeRun;
|
|
2222
|
+
items: LoopJudgeWorkItem[];
|
|
2223
|
+
remaining: number;
|
|
2224
|
+
}
|
|
2225
|
+
/** One scored conversation going back. */
|
|
2226
|
+
export interface LoopJudgeVerdict {
|
|
2227
|
+
trace_id: string;
|
|
2228
|
+
/** One score per dimension the rubric asked for, each between 0 and 1. */
|
|
2229
|
+
scores?: Record<string, number>;
|
|
2230
|
+
/** Your own combination of them. Left out, the plain mean is used. */
|
|
2231
|
+
overall?: number;
|
|
2232
|
+
reason?: string;
|
|
2233
|
+
/** Only stored if the judge was created with `write_gold`. */
|
|
2234
|
+
gold?: string;
|
|
2235
|
+
/** Mark an item you could not score, instead of dropping it silently. */
|
|
2236
|
+
error?: string;
|
|
2237
|
+
}
|
|
2238
|
+
export interface LoopJudgeVerdictResult {
|
|
2239
|
+
run: LoopJudgeRun;
|
|
2240
|
+
recorded: number;
|
|
2241
|
+
failed: number;
|
|
2242
|
+
rejected: Array<{
|
|
2243
|
+
trace_id: string;
|
|
2244
|
+
error: string;
|
|
2245
|
+
}>;
|
|
2246
|
+
}
|
|
2145
2247
|
export interface LoopStats {
|
|
2146
2248
|
traces: number;
|
|
2147
2249
|
signalled_traces: number;
|
|
@@ -2205,6 +2307,12 @@ export interface LoopGraderParams {
|
|
|
2205
2307
|
enabled?: boolean;
|
|
2206
2308
|
/** Limit the rule to one capture source. Omit to apply it everywhere. */
|
|
2207
2309
|
deployment_id?: string;
|
|
2310
|
+
/**
|
|
2311
|
+
* Aim the rule at one slice of the corpus. A master group covers its
|
|
2312
|
+
* children. Outside that slice the rule is NOT APPLIED, rather than failed,
|
|
2313
|
+
* so it stays out of the score entirely. Omit to apply it everywhere.
|
|
2314
|
+
*/
|
|
2315
|
+
label?: string;
|
|
2208
2316
|
}
|
|
2209
2317
|
export interface LoopGrader extends LoopGraderParams {
|
|
2210
2318
|
id: string;
|