runbios-sdk 0.2.1-dev.106 → 0.2.1-dev.108

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/dist/index.d.ts CHANGED
@@ -35,7 +35,7 @@ import { Loop } from './resources/loop.js';
35
35
  * SDK version. Sent as part of the User-Agent header.
36
36
  * Must match package.json "version" -- enforced by a contract test.
37
37
  */
38
- export declare const VERSION = "0.2.1-dev.106";
38
+ export declare const VERSION = "0.2.1-dev.108";
39
39
  export declare class RunBiOS {
40
40
  /** Search models, fetch configs, check adapter compatibility. */
41
41
  readonly models: Models;
package/dist/index.js CHANGED
@@ -35,7 +35,7 @@ import { Loop } from './resources/loop.js';
35
35
  * SDK version. Sent as part of the User-Agent header.
36
36
  * Must match package.json "version" -- enforced by a contract test.
37
37
  */
38
- export const VERSION = '0.2.1-dev.106';
38
+ export const VERSION = '0.2.1-dev.108';
39
39
  export class RunBiOS {
40
40
  /** Search models, fetch configs, check adapter compatibility. */
41
41
  models;
@@ -1,5 +1,5 @@
1
1
  import type { HttpClient } from '../client.js';
2
- import type { LoopTrace, LoopTraceListResponse, LoopTraceListParams, LoopCaptureParams, LoopCaptureResult, LoopSignal, LoopSignalParams, LoopCandidate, LoopCandidateParams, LoopDataset, LoopDatasetListParams, LoopDatasetItem, LoopDatasetCreateParams, LoopConfig, LoopConfigParams, LoopStats } from '../types.js';
2
+ import type { LoopTrace, LoopTraceListResponse, LoopTraceListParams, LoopCaptureParams, LoopCaptureResult, LoopSignal, LoopSignalParams, LoopCandidate, LoopCandidateParams, LoopGrader, LoopGraderParams, LoopGradeResult, LoopDataset, LoopDatasetListParams, LoopDatasetItem, LoopDatasetCreateParams, LoopConfig, LoopConfigParams, LoopStats } from '../types.js';
3
3
  /**
4
4
  * The Conscious Loop -- capture what your model was asked and answered, record
5
5
  * whether it was right, and turn those judgements into training data.
@@ -88,6 +88,42 @@ export declare class Loop {
88
88
  * behavioural hint.
89
89
  */
90
90
  signal(traceId: string, params: LoopSignalParams): Promise<LoopSignal>;
91
+ /**
92
+ * Thumbs up: this answer was good. Shorthand for `signal(id, { verdict:
93
+ * 'accepted' })` — there is one stored vocabulary behind both, so a thumbs-up
94
+ * and an "accepted" can never disagree in your corpus.
95
+ */
96
+ upvote(traceId: string, reason?: string): Promise<LoopSignal>;
97
+ /**
98
+ * Thumbs down: this answer was bad, and you have nothing better to offer.
99
+ *
100
+ * On its own this REMOVES the answer from training rather than teaching
101
+ * anything — there is no better version to learn. If you know what it should
102
+ * have said, `correct()` is worth far more.
103
+ */
104
+ downvote(traceId: string, reason?: string): Promise<LoopSignal>;
105
+ /**
106
+ * The model was wrong; here is what it should have said.
107
+ *
108
+ * The most valuable feedback there is, because it produces BOTH halves of a
109
+ * preference pair from one action: the model learns your answer and learns to
110
+ * avoid its own.
111
+ */
112
+ correct(traceId: string, answer: string, reason?: string): Promise<LoopSignal>;
113
+ /**
114
+ * Record the GOLD answer for this question — the reference, regardless of
115
+ * what the model happened to say.
116
+ *
117
+ * Different from `correct()` in a way that matters: a correction asserts the
118
+ * model was wrong, so the original becomes the rejected half of a pair. A gold
119
+ * answer asserts nothing about the model, so if it MATCHES what was said, no
120
+ * preference pair is invented — but SFT still learns it.
121
+ *
122
+ * It is the most reusable thing you can record: it trains SFT, forms a DPO
123
+ * pair when it differs from the answer, and stands in as the GRPO reference
124
+ * when you have not supplied a separate ground truth.
125
+ */
126
+ gold(traceId: string, answer: string, reason?: string): Promise<LoopSignal>;
91
127
  /** Every verdict on one conversation, oldest first. */
92
128
  listSignals(traceId: string): Promise<LoopSignal[]>;
93
129
  /**
@@ -172,6 +208,47 @@ export declare class Loop {
172
208
  deleted: boolean;
173
209
  dataset_id: string;
174
210
  }>;
211
+ /**
212
+ * Write a rule that scores answers without a person.
213
+ *
214
+ * Human review is the most trustworthy feedback and the least available. A
215
+ * grader is written once and applied to every answer afterwards: did it
216
+ * contain the required phrase, did it parse as the schema you asked for, is
217
+ * the number within tolerance of the known answer, did it call the function
218
+ * it should have.
219
+ *
220
+ * Every kind here is DETERMINISTIC -- no model call, no network. That is why
221
+ * a verifier outranks a judge when they disagree: it cannot be flattered and
222
+ * it cannot drift between runs.
223
+ *
224
+ * `weight` is the multiplier: a criterion that matters twice as much gets
225
+ * twice the weight, and the combined score is the weighted mean over the
226
+ * rules that actually applied.
227
+ *
228
+ * A caution worth knowing before you write a set: a rule made only of
229
+ * `forbidden` phrases is satisfied VACUOUSLY by an answer that says nothing.
230
+ * Pair it with a `required` phrase, or you are rewarding silence.
231
+ */
232
+ createGrader(params: LoopGraderParams): Promise<LoopGrader>;
233
+ /** Every rule this workspace has written. */
234
+ listGraders(): Promise<LoopGrader[]>;
235
+ deleteGrader(id: string): Promise<{
236
+ deleted: boolean;
237
+ grader_id: string;
238
+ }>;
239
+ /**
240
+ * Apply this workspace's rules to one captured answer AND to every
241
+ * alternative sampled for it.
242
+ *
243
+ * Grading both in one pass is the point: scoring only the original gives a
244
+ * verdict, while scoring the samples as well gives the preference pair. Sample
245
+ * your model, call this, and you have DPO data with nobody reading anything.
246
+ *
247
+ * A run where no rule could apply writes NOTHING -- no verdict, no scores.
248
+ * Recording a zero that no rule produced would poison curation with a
249
+ * judgement nobody reached, so `applied: 0` is reported instead.
250
+ */
251
+ grade(traceId: string): Promise<LoopGradeResult>;
175
252
  /** Every source this workspace has configured. */
176
253
  listConfigs(): Promise<LoopConfig[]>;
177
254
  /**
@@ -116,6 +116,50 @@ export class Loop {
116
116
  const res = await this._http.fetchPost(`/api/loop/traces/${encodeURIComponent(traceId)}/signals`, params);
117
117
  return res.signal;
118
118
  }
119
+ /**
120
+ * Thumbs up: this answer was good. Shorthand for `signal(id, { verdict:
121
+ * 'accepted' })` — there is one stored vocabulary behind both, so a thumbs-up
122
+ * and an "accepted" can never disagree in your corpus.
123
+ */
124
+ async upvote(traceId, reason) {
125
+ return this.signal(traceId, { verdict: 'accepted', reason });
126
+ }
127
+ /**
128
+ * Thumbs down: this answer was bad, and you have nothing better to offer.
129
+ *
130
+ * On its own this REMOVES the answer from training rather than teaching
131
+ * anything — there is no better version to learn. If you know what it should
132
+ * have said, `correct()` is worth far more.
133
+ */
134
+ async downvote(traceId, reason) {
135
+ return this.signal(traceId, { verdict: 'rejected', reason });
136
+ }
137
+ /**
138
+ * The model was wrong; here is what it should have said.
139
+ *
140
+ * The most valuable feedback there is, because it produces BOTH halves of a
141
+ * preference pair from one action: the model learns your answer and learns to
142
+ * avoid its own.
143
+ */
144
+ async correct(traceId, answer, reason) {
145
+ return this.signal(traceId, { verdict: 'edited', correction: answer, reason });
146
+ }
147
+ /**
148
+ * Record the GOLD answer for this question — the reference, regardless of
149
+ * what the model happened to say.
150
+ *
151
+ * Different from `correct()` in a way that matters: a correction asserts the
152
+ * model was wrong, so the original becomes the rejected half of a pair. A gold
153
+ * answer asserts nothing about the model, so if it MATCHES what was said, no
154
+ * preference pair is invented — but SFT still learns it.
155
+ *
156
+ * It is the most reusable thing you can record: it trains SFT, forms a DPO
157
+ * pair when it differs from the answer, and stands in as the GRPO reference
158
+ * when you have not supplied a separate ground truth.
159
+ */
160
+ async gold(traceId, answer, reason) {
161
+ return this.signal(traceId, { verdict: 'gold', correction: answer, reason });
162
+ }
119
163
  /** Every verdict on one conversation, oldest first. */
120
164
  async listSignals(traceId) {
121
165
  const res = await this._http.fetchGet(`/api/loop/traces/${encodeURIComponent(traceId)}/signals`);
@@ -236,6 +280,55 @@ export class Loop {
236
280
  async deleteDataset(id) {
237
281
  return this._http.fetchDelete(`/api/loop/datasets/${encodeURIComponent(id)}`);
238
282
  }
283
+ // ── graders ───────────────────────────────────────────────────────────
284
+ /**
285
+ * Write a rule that scores answers without a person.
286
+ *
287
+ * Human review is the most trustworthy feedback and the least available. A
288
+ * grader is written once and applied to every answer afterwards: did it
289
+ * contain the required phrase, did it parse as the schema you asked for, is
290
+ * the number within tolerance of the known answer, did it call the function
291
+ * it should have.
292
+ *
293
+ * Every kind here is DETERMINISTIC -- no model call, no network. That is why
294
+ * a verifier outranks a judge when they disagree: it cannot be flattered and
295
+ * it cannot drift between runs.
296
+ *
297
+ * `weight` is the multiplier: a criterion that matters twice as much gets
298
+ * twice the weight, and the combined score is the weighted mean over the
299
+ * rules that actually applied.
300
+ *
301
+ * A caution worth knowing before you write a set: a rule made only of
302
+ * `forbidden` phrases is satisfied VACUOUSLY by an answer that says nothing.
303
+ * Pair it with a `required` phrase, or you are rewarding silence.
304
+ */
305
+ async createGrader(params) {
306
+ const res = await this._http.fetchPost('/api/loop/graders', params);
307
+ return res.grader;
308
+ }
309
+ /** Every rule this workspace has written. */
310
+ async listGraders() {
311
+ const res = await this._http.fetchGet('/api/loop/graders');
312
+ return res.graders;
313
+ }
314
+ async deleteGrader(id) {
315
+ return this._http.fetchDelete(`/api/loop/graders/${encodeURIComponent(id)}`);
316
+ }
317
+ /**
318
+ * Apply this workspace's rules to one captured answer AND to every
319
+ * alternative sampled for it.
320
+ *
321
+ * Grading both in one pass is the point: scoring only the original gives a
322
+ * verdict, while scoring the samples as well gives the preference pair. Sample
323
+ * your model, call this, and you have DPO data with nobody reading anything.
324
+ *
325
+ * A run where no rule could apply writes NOTHING -- no verdict, no scores.
326
+ * Recording a zero that no rule produced would poison curation with a
327
+ * judgement nobody reached, so `applied: 0` is reported instead.
328
+ */
329
+ async grade(traceId) {
330
+ return this._http.fetchPost(`/api/loop/traces/${encodeURIComponent(traceId)}/grade`);
331
+ }
239
332
  // ── capture settings ──────────────────────────────────────────────────
240
333
  /** Every source this workspace has configured. */
241
334
  async listConfigs() {
package/dist/types.d.ts CHANGED
@@ -1960,7 +1960,15 @@ export type LoopMethod = 'sft' | 'dpo' | 'grpo';
1960
1960
  /** Who judged an answer. */
1961
1961
  export type LoopSource = 'human' | 'verifier' | 'judge' | 'behavioural';
1962
1962
  /** The judgement itself. */
1963
- export type LoopVerdict = 'accepted' | 'rejected' | 'edited' | 'scored';
1963
+ /**
1964
+ * The judgement recorded on an answer.
1965
+ *
1966
+ * `accepted` / `rejected` are the thumbs (see `upvote` / `downvote`).
1967
+ * `edited` says the model was WRONG and carries the better answer.
1968
+ * `gold` records the reference answer for the question, making no claim about
1969
+ * whether the model was right — which is why it is separate from `edited`.
1970
+ */
1971
+ export type LoopVerdict = 'accepted' | 'rejected' | 'edited' | 'scored' | 'gold';
1964
1972
  /** A function an assistant turn asked to invoke. `arguments` is JSON-encoded. */
1965
1973
  export interface LoopToolCall {
1966
1974
  id?: string;
@@ -2167,3 +2175,65 @@ export interface LoopCandidate {
2167
2175
  metadata: Record<string, unknown>;
2168
2176
  created_at: string;
2169
2177
  }
2178
+ /** The deterministic checks a grader can perform. */
2179
+ export type LoopGraderKind = 'exact_match' | 'contains' | 'regex' | 'json_valid' | 'numeric' | 'tool_called';
2180
+ export interface LoopGraderConfig {
2181
+ expected?: string;
2182
+ /** Every phrase that must appear. */
2183
+ required?: string[];
2184
+ /** Phrases that must not. On their own these are passed by silence. */
2185
+ forbidden?: string[];
2186
+ pattern?: string;
2187
+ /** Keys the answer must carry, for `json_valid`. */
2188
+ keys?: string[];
2189
+ /** How far from `expected` still counts, for `numeric`. */
2190
+ tolerance?: number;
2191
+ /** The function that must have been called, for `tool_called`. */
2192
+ function?: string;
2193
+ case_sensitive?: boolean;
2194
+ }
2195
+ export interface LoopGraderParams {
2196
+ name: string;
2197
+ kind: LoopGraderKind;
2198
+ config: LoopGraderConfig;
2199
+ /** The multiplier. Twice as important, twice the weight. Must exceed zero. */
2200
+ weight?: number;
2201
+ enabled?: boolean;
2202
+ /** Limit the rule to one capture source. Omit to apply it everywhere. */
2203
+ deployment_id?: string;
2204
+ }
2205
+ export interface LoopGrader extends LoopGraderParams {
2206
+ id: string;
2207
+ workspace_id: string;
2208
+ weight: number;
2209
+ enabled: boolean;
2210
+ created_by: string | null;
2211
+ created_at: string;
2212
+ updated_at: string;
2213
+ }
2214
+ export interface LoopGraderResult {
2215
+ grader_id: string;
2216
+ name: string;
2217
+ kind: string;
2218
+ weight: number;
2219
+ passed: boolean;
2220
+ score: number;
2221
+ /** What happened, in words -- "failed" alone sends you to read the rule. */
2222
+ detail: string;
2223
+ /** The rule had no opinion. Excluded from the score rather than counted 0. */
2224
+ skipped: boolean;
2225
+ }
2226
+ export interface LoopGradeReport {
2227
+ /** Weighted mean over the rules that applied, 0 to 1. */
2228
+ score: number;
2229
+ results: LoopGraderResult[];
2230
+ /** The denominator. 1.0 from one rule is not 1.0 from six. */
2231
+ applied: number;
2232
+ skipped: number;
2233
+ }
2234
+ export interface LoopGradeResult {
2235
+ report: LoopGradeReport;
2236
+ /** Null when no rule applied, because no verdict was invented. */
2237
+ signal_id: string | null;
2238
+ candidates_graded: number;
2239
+ }
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "runbios-sdk",
3
- "version": "0.2.1-dev.106",
3
+ "version": "0.2.1-dev.108",
4
4
  "description": "Official TypeScript SDK for the Run BiOS training and deployment platform API",
5
5
  "type": "module",
6
6
  "main": "./dist/index.js",