runbios-sdk 0.2.1-dev.109 → 0.2.1-dev.110

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/dist/index.d.ts CHANGED
@@ -35,7 +35,7 @@ import { Loop } from './resources/loop.js';
35
35
  * SDK version. Sent as part of the User-Agent header.
36
36
  * Must match package.json "version" -- enforced by a contract test.
37
37
  */
38
- export declare const VERSION = "0.2.1-dev.109";
38
+ export declare const VERSION = "0.2.1-dev.110";
39
39
  export declare class RunBiOS {
40
40
  /** Search models, fetch configs, check adapter compatibility. */
41
41
  readonly models: Models;
package/dist/index.js CHANGED
@@ -35,7 +35,7 @@ import { Loop } from './resources/loop.js';
35
35
  * SDK version. Sent as part of the User-Agent header.
36
36
  * Must match package.json "version" -- enforced by a contract test.
37
37
  */
38
- export const VERSION = '0.2.1-dev.109';
38
+ export const VERSION = '0.2.1-dev.110';
39
39
  export class RunBiOS {
40
40
  /** Search models, fetch configs, check adapter compatibility. */
41
41
  models;
@@ -1,5 +1,5 @@
1
1
  import type { HttpClient } from '../client.js';
2
- import type { LoopTrace, LoopTraceListResponse, LoopTraceListParams, LoopCaptureParams, LoopCaptureResult, LoopSignal, LoopSignalParams, LoopCandidate, LoopCandidateParams, LoopGrader, LoopGraderParams, LoopLabel, LoopLabelCount, LoopGradeResult, LoopDataset, LoopDatasetListParams, LoopDatasetItem, LoopDatasetCreateParams, LoopConfig, LoopConfigParams, LoopStats } from '../types.js';
2
+ import type { LoopTrace, LoopTraceListResponse, LoopTraceListParams, LoopCaptureParams, LoopCaptureResult, LoopSignal, LoopSignalParams, LoopCandidate, LoopCandidateParams, LoopGrader, LoopGraderParams, LoopLabel, LoopLabelCount, LoopGradeResult, LoopDataset, LoopDatasetListParams, LoopDatasetItem, LoopDatasetCreateParams, LoopConfig, LoopConfigParams, LoopStats, LoopJudge, LoopJudgeParams, LoopJudgeRun, LoopJudgeWork, LoopJudgeVerdict, LoopJudgeVerdictResult } from '../types.js';
3
3
  /**
4
4
  * The Conscious Loop -- capture what your model was asked and answered, record
5
5
  * whether it was right, and turn those judgements into training data.
@@ -295,4 +295,44 @@ export declare class Loop {
295
295
  * one row while a set is built, so the finished count can be lower.
296
296
  */
297
297
  stats(): Promise<LoopStats>;
298
+ /** Write a rubric: what makes an answer good here, and what to score. */
299
+ createJudge(params: LoopJudgeParams): Promise<LoopJudge>;
300
+ listJudges(): Promise<LoopJudge[]>;
301
+ /**
302
+ * Rewrite a rubric. Runs already recorded keep the instructions they used:
303
+ * a verdict has to keep meaning what it meant when it was given.
304
+ */
305
+ updateJudge(judgeId: string, params: LoopJudgeParams): Promise<LoopJudge>;
306
+ /**
307
+ * Retire a rubric. Its runs go with it; the verdicts they wrote stay. A score
308
+ * is evidence about an answer, and it does not stop being true because the
309
+ * rubric was retired.
310
+ */
311
+ deleteJudge(judgeId: string): Promise<void>;
312
+ /**
313
+ * Open a run: select the conversations now, and freeze the rubric onto them.
314
+ *
315
+ * The selection is fixed at this moment on purpose. A run whose selection is
316
+ * a live query silently grows as conversations arrive, so "we scored the
317
+ * refunds slice" becomes a claim about a set that no longer exists.
318
+ */
319
+ startRun(judgeId: string, limit?: number): Promise<LoopJudgeRun>;
320
+ listRuns(judgeId?: string): Promise<LoopJudgeRun[]>;
321
+ getRun(runId: string): Promise<LoopJudgeRun>;
322
+ /**
323
+ * Take the next conversations to score, each rendered into a ready-to-send
324
+ * prompt. Nothing is marked taken, so a caller that dies half way loses
325
+ * nothing: ask again and the same items come back.
326
+ */
327
+ takeWork(runId: string, limit?: number): Promise<LoopJudgeWork>;
328
+ /**
329
+ * Hand the scores back. A verdict for a conversation outside this run, or
330
+ * scoring something the rubric never asked for, is refused and reported in
331
+ * `rejected` rather than silently dropped.
332
+ *
333
+ * Pass `finish` to close a run you have decided to stop early. Without it an
334
+ * unfinished run stays open, which is the honest state for work that was
335
+ * abandoned rather than completed.
336
+ */
337
+ postVerdicts(runId: string, verdicts: LoopJudgeVerdict[], finish?: boolean): Promise<LoopJudgeVerdictResult>;
298
338
  }
@@ -394,4 +394,82 @@ export class Loop {
394
394
  const res = await this._http.fetchGet('/api/loop/stats');
395
395
  return res.stats;
396
396
  }
397
+ // ── judges ────────────────────────────────────────────────────────────
398
+ //
399
+ // A grader is a rule: deterministic, cheap, and blind to anything it was not
400
+ // told to look for. A judge is a rubric handed to a model, which is the only
401
+ // thing that can answer "was this answer actually helpful".
402
+ //
403
+ // YOU run the model. This service stores raw prompts and completions and
404
+ // holds them with no outbound credentials at all, which is most of the reason
405
+ // it is safe to store them there -- so it hands the work out instead. Open a
406
+ // run, take the batch, send each `prompt` to whatever model you like, and
407
+ // post the scores back.
408
+ /** Write a rubric: what makes an answer good here, and what to score. */
409
+ async createJudge(params) {
410
+ const res = await this._http.fetchPost('/api/loop/judges', params);
411
+ return res.judge;
412
+ }
413
+ async listJudges() {
414
+ const res = await this._http.fetchGet('/api/loop/judges');
415
+ return res.judges;
416
+ }
417
+ /**
418
+ * Rewrite a rubric. Runs already recorded keep the instructions they used:
419
+ * a verdict has to keep meaning what it meant when it was given.
420
+ */
421
+ async updateJudge(judgeId, params) {
422
+ const res = await this._http.fetchPut(`/api/loop/judges/${encodeURIComponent(judgeId)}`, params);
423
+ return res.judge;
424
+ }
425
+ /**
426
+ * Retire a rubric. Its runs go with it; the verdicts they wrote stay. A score
427
+ * is evidence about an answer, and it does not stop being true because the
428
+ * rubric was retired.
429
+ */
430
+ async deleteJudge(judgeId) {
431
+ await this._http.fetchDelete(`/api/loop/judges/${encodeURIComponent(judgeId)}`);
432
+ }
433
+ /**
434
+ * Open a run: select the conversations now, and freeze the rubric onto them.
435
+ *
436
+ * The selection is fixed at this moment on purpose. A run whose selection is
437
+ * a live query silently grows as conversations arrive, so "we scored the
438
+ * refunds slice" becomes a claim about a set that no longer exists.
439
+ */
440
+ async startRun(judgeId, limit) {
441
+ const res = await this._http.fetchPost(`/api/loop/judges/${encodeURIComponent(judgeId)}/runs`, limit ? { limit } : {});
442
+ return res.run;
443
+ }
444
+ async listRuns(judgeId) {
445
+ const path = judgeId
446
+ ? `/api/loop/judges/${encodeURIComponent(judgeId)}/runs`
447
+ : '/api/loop/runs';
448
+ const res = await this._http.fetchGet(path);
449
+ return res.runs;
450
+ }
451
+ async getRun(runId) {
452
+ const res = await this._http.fetchGet(`/api/loop/runs/${encodeURIComponent(runId)}`);
453
+ return res.run;
454
+ }
455
+ /**
456
+ * Take the next conversations to score, each rendered into a ready-to-send
457
+ * prompt. Nothing is marked taken, so a caller that dies half way loses
458
+ * nothing: ask again and the same items come back.
459
+ */
460
+ async takeWork(runId, limit = 20) {
461
+ return this._http.fetchGet(`/api/loop/runs/${encodeURIComponent(runId)}/work?limit=${limit}`);
462
+ }
463
+ /**
464
+ * Hand the scores back. A verdict for a conversation outside this run, or
465
+ * scoring something the rubric never asked for, is refused and reported in
466
+ * `rejected` rather than silently dropped.
467
+ *
468
+ * Pass `finish` to close a run you have decided to stop early. Without it an
469
+ * unfinished run stays open, which is the honest state for work that was
470
+ * abandoned rather than completed.
471
+ */
472
+ async postVerdicts(runId, verdicts, finish = false) {
473
+ return this._http.fetchPost(`/api/loop/runs/${encodeURIComponent(runId)}/verdicts`, { verdicts, finish });
474
+ }
397
475
  }
package/dist/types.d.ts CHANGED
@@ -2142,6 +2142,108 @@ export interface LoopConfigParams {
2142
2142
  /** Deterministic per conversation, so turns are never split. 0 < rate <= 1. */
2143
2143
  sample_rate?: number;
2144
2144
  }
2145
+ /** One thing a judge scores, separately from the others. */
2146
+ export interface LoopJudgeDimension {
2147
+ key: string;
2148
+ description?: string;
2149
+ }
2150
+ /** Which slice of the corpus a judge is responsible for. */
2151
+ export interface LoopJudgeSelection {
2152
+ label?: string;
2153
+ deployment_id?: string;
2154
+ from?: string;
2155
+ to?: string;
2156
+ sample?: number;
2157
+ /** Skip conversations that already carry a judge verdict. */
2158
+ only_unscored?: boolean;
2159
+ }
2160
+ export interface LoopJudge {
2161
+ id: string;
2162
+ workspace_id: string;
2163
+ name: string;
2164
+ instructions: string;
2165
+ dimensions: LoopJudgeDimension[];
2166
+ selection: LoopJudgeSelection;
2167
+ model: string | null;
2168
+ write_gold: boolean;
2169
+ enabled: boolean;
2170
+ created_by: string | null;
2171
+ created_at: string;
2172
+ updated_at: string;
2173
+ }
2174
+ export interface LoopJudgeParams {
2175
+ name: string;
2176
+ instructions: string;
2177
+ dimensions: LoopJudgeDimension[];
2178
+ selection?: LoopJudgeSelection;
2179
+ model?: string;
2180
+ /**
2181
+ * Let this judge write the answer that should have been given, not just a
2182
+ * score. Off by default: a reference answer written by a model and then
2183
+ * trained on is distillation, which is a decision you make deliberately.
2184
+ */
2185
+ write_gold?: boolean;
2186
+ enabled?: boolean;
2187
+ }
2188
+ /** One pass of one rubric over one slice. */
2189
+ export interface LoopJudgeRun {
2190
+ id: string;
2191
+ judge_id: string;
2192
+ judge_name?: string;
2193
+ status: 'open' | 'done';
2194
+ instructions: string;
2195
+ dimensions: LoopJudgeDimension[];
2196
+ model: string | null;
2197
+ selected: number;
2198
+ scored: number;
2199
+ failed: number;
2200
+ created_by: string | null;
2201
+ created_at: string;
2202
+ finished_at: string | null;
2203
+ }
2204
+ /**
2205
+ * One conversation to score, already rendered into the prompt to send.
2206
+ *
2207
+ * Send `prompt` as-is. Assembling it yourself is how two callers end up giving
2208
+ * the same rubric different instructions, and two judges given different
2209
+ * instructions are not one judge.
2210
+ */
2211
+ export interface LoopJudgeWorkItem {
2212
+ trace_id: string;
2213
+ messages: Array<{
2214
+ role: string;
2215
+ content?: string;
2216
+ }>;
2217
+ answer: string;
2218
+ prompt: string;
2219
+ }
2220
+ export interface LoopJudgeWork {
2221
+ run: LoopJudgeRun;
2222
+ items: LoopJudgeWorkItem[];
2223
+ remaining: number;
2224
+ }
2225
+ /** One scored conversation going back. */
2226
+ export interface LoopJudgeVerdict {
2227
+ trace_id: string;
2228
+ /** One score per dimension the rubric asked for, each between 0 and 1. */
2229
+ scores?: Record<string, number>;
2230
+ /** Your own combination of them. Left out, the plain mean is used. */
2231
+ overall?: number;
2232
+ reason?: string;
2233
+ /** Only stored if the judge was created with `write_gold`. */
2234
+ gold?: string;
2235
+ /** Mark an item you could not score, instead of dropping it silently. */
2236
+ error?: string;
2237
+ }
2238
+ export interface LoopJudgeVerdictResult {
2239
+ run: LoopJudgeRun;
2240
+ recorded: number;
2241
+ failed: number;
2242
+ rejected: Array<{
2243
+ trace_id: string;
2244
+ error: string;
2245
+ }>;
2246
+ }
2145
2247
  export interface LoopStats {
2146
2248
  traces: number;
2147
2249
  signalled_traces: number;
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "runbios-sdk",
3
- "version": "0.2.1-dev.109",
3
+ "version": "0.2.1-dev.110",
4
4
  "description": "Official TypeScript SDK for the Run BiOS training and deployment platform API",
5
5
  "type": "module",
6
6
  "main": "./dist/index.js",