runbios-sdk 0.2.1-dev.107 → 0.2.1-dev.109

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/dist/index.d.ts CHANGED
@@ -35,7 +35,7 @@ import { Loop } from './resources/loop.js';
35
35
  * SDK version. Sent as part of the User-Agent header.
36
36
  * Must match package.json "version" -- enforced by a contract test.
37
37
  */
38
- export declare const VERSION = "0.2.1-dev.107";
38
+ export declare const VERSION = "0.2.1-dev.109";
39
39
  export declare class RunBiOS {
40
40
  /** Search models, fetch configs, check adapter compatibility. */
41
41
  readonly models: Models;
package/dist/index.js CHANGED
@@ -35,7 +35,7 @@ import { Loop } from './resources/loop.js';
35
35
  * SDK version. Sent as part of the User-Agent header.
36
36
  * Must match package.json "version" -- enforced by a contract test.
37
37
  */
38
- export const VERSION = '0.2.1-dev.107';
38
+ export const VERSION = '0.2.1-dev.109';
39
39
  export class RunBiOS {
40
40
  /** Search models, fetch configs, check adapter compatibility. */
41
41
  models;
@@ -1,5 +1,5 @@
1
1
  import type { HttpClient } from '../client.js';
2
- import type { LoopTrace, LoopTraceListResponse, LoopTraceListParams, LoopCaptureParams, LoopCaptureResult, LoopSignal, LoopSignalParams, LoopCandidate, LoopCandidateParams, LoopGrader, LoopGraderParams, LoopGradeResult, LoopDataset, LoopDatasetListParams, LoopDatasetItem, LoopDatasetCreateParams, LoopConfig, LoopConfigParams, LoopStats } from '../types.js';
2
+ import type { LoopTrace, LoopTraceListResponse, LoopTraceListParams, LoopCaptureParams, LoopCaptureResult, LoopSignal, LoopSignalParams, LoopCandidate, LoopCandidateParams, LoopGrader, LoopGraderParams, LoopLabel, LoopLabelCount, LoopGradeResult, LoopDataset, LoopDatasetListParams, LoopDatasetItem, LoopDatasetCreateParams, LoopConfig, LoopConfigParams, LoopStats } from '../types.js';
3
3
  /**
4
4
  * The Conscious Loop -- capture what your model was asked and answered, record
5
5
  * whether it was right, and turn those judgements into training data.
@@ -88,6 +88,42 @@ export declare class Loop {
88
88
  * behavioural hint.
89
89
  */
90
90
  signal(traceId: string, params: LoopSignalParams): Promise<LoopSignal>;
91
+ /**
92
+ * Thumbs up: this answer was good. Shorthand for `signal(id, { verdict:
93
+ * 'accepted' })` — there is one stored vocabulary behind both, so a thumbs-up
94
+ * and an "accepted" can never disagree in your corpus.
95
+ */
96
+ upvote(traceId: string, reason?: string): Promise<LoopSignal>;
97
+ /**
98
+ * Thumbs down: this answer was bad, and you have nothing better to offer.
99
+ *
100
+ * On its own this REMOVES the answer from training rather than teaching
101
+ * anything — there is no better version to learn. If you know what it should
102
+ * have said, `correct()` is worth far more.
103
+ */
104
+ downvote(traceId: string, reason?: string): Promise<LoopSignal>;
105
+ /**
106
+ * The model was wrong; here is what it should have said.
107
+ *
108
+ * The most valuable feedback there is, because it produces BOTH halves of a
109
+ * preference pair from one action: the model learns your answer and learns to
110
+ * avoid its own.
111
+ */
112
+ correct(traceId: string, answer: string, reason?: string): Promise<LoopSignal>;
113
+ /**
114
+ * Record the GOLD answer for this question — the reference, regardless of
115
+ * what the model happened to say.
116
+ *
117
+ * Different from `correct()` in a way that matters: a correction asserts the
118
+ * model was wrong, so the original becomes the rejected half of a pair. A gold
119
+ * answer asserts nothing about the model, so if it MATCHES what was said, no
120
+ * preference pair is invented — but SFT still learns it.
121
+ *
122
+ * It is the most reusable thing you can record: it trains SFT, forms a DPO
123
+ * pair when it differs from the answer, and stands in as the GRPO reference
124
+ * when you have not supplied a separate ground truth.
125
+ */
126
+ gold(traceId: string, answer: string, reason?: string): Promise<LoopSignal>;
91
127
  /** Every verdict on one conversation, oldest first. */
92
128
  listSignals(traceId: string): Promise<LoopSignal[]>;
93
129
  /**
@@ -121,6 +157,27 @@ export declare class Loop {
121
157
  addCandidate(traceId: string, params: LoopCandidateParams): Promise<LoopCandidate>;
122
158
  /** Every alternative answer recorded for one prompt, oldest first. */
123
159
  listCandidates(traceId: string): Promise<LoopCandidate[]>;
160
+ /**
161
+ * Label a conversation, so it can be selected later.
162
+ *
163
+ * An average over everything is the least useful thing to train on. A model
164
+ * weak at refunds is fixed with refund examples, and you can only select
165
+ * those if you said so when the conversation arrived.
166
+ *
167
+ * A label may name a `parent`, which is what makes a MASTER GROUP: labelling
168
+ * something `refunds` with parent `billing` makes it selectable as either,
169
+ * and selecting `billing` later gathers every child without you maintaining
170
+ * a list of them.
171
+ *
172
+ * Labels are lowercased and trimmed, so `Refunds` and `refunds` are one label.
173
+ */
174
+ label(traceId: string, labels: string[], parent?: string): Promise<LoopLabel[]>;
175
+ unlabel(traceId: string, label: string): Promise<{
176
+ deleted: boolean;
177
+ label: string;
178
+ }>;
179
+ /** Every label in the workspace, with how many conversations carry it. */
180
+ listLabels(): Promise<LoopLabelCount[]>;
124
181
  /**
125
182
  * Build a training set from the feedback recorded so far.
126
183
  *
@@ -116,6 +116,50 @@ export class Loop {
116
116
  const res = await this._http.fetchPost(`/api/loop/traces/${encodeURIComponent(traceId)}/signals`, params);
117
117
  return res.signal;
118
118
  }
119
+ /**
120
+ * Thumbs up: this answer was good. Shorthand for `signal(id, { verdict:
121
+ * 'accepted' })` — there is one stored vocabulary behind both, so a thumbs-up
122
+ * and an "accepted" can never disagree in your corpus.
123
+ */
124
+ async upvote(traceId, reason) {
125
+ return this.signal(traceId, { verdict: 'accepted', reason });
126
+ }
127
+ /**
128
+ * Thumbs down: this answer was bad, and you have nothing better to offer.
129
+ *
130
+ * On its own this REMOVES the answer from training rather than teaching
131
+ * anything — there is no better version to learn. If you know what it should
132
+ * have said, `correct()` is worth far more.
133
+ */
134
+ async downvote(traceId, reason) {
135
+ return this.signal(traceId, { verdict: 'rejected', reason });
136
+ }
137
+ /**
138
+ * The model was wrong; here is what it should have said.
139
+ *
140
+ * The most valuable feedback there is, because it produces BOTH halves of a
141
+ * preference pair from one action: the model learns your answer and learns to
142
+ * avoid its own.
143
+ */
144
+ async correct(traceId, answer, reason) {
145
+ return this.signal(traceId, { verdict: 'edited', correction: answer, reason });
146
+ }
147
+ /**
148
+ * Record the GOLD answer for this question — the reference, regardless of
149
+ * what the model happened to say.
150
+ *
151
+ * Different from `correct()` in a way that matters: a correction asserts the
152
+ * model was wrong, so the original becomes the rejected half of a pair. A gold
153
+ * answer asserts nothing about the model, so if it MATCHES what was said, no
154
+ * preference pair is invented — but SFT still learns it.
155
+ *
156
+ * It is the most reusable thing you can record: it trains SFT, forms a DPO
157
+ * pair when it differs from the answer, and stands in as the GRPO reference
158
+ * when you have not supplied a separate ground truth.
159
+ */
160
+ async gold(traceId, answer, reason) {
161
+ return this.signal(traceId, { verdict: 'gold', correction: answer, reason });
162
+ }
119
163
  /** Every verdict on one conversation, oldest first. */
120
164
  async listSignals(traceId) {
121
165
  const res = await this._http.fetchGet(`/api/loop/traces/${encodeURIComponent(traceId)}/signals`);
@@ -159,6 +203,33 @@ export class Loop {
159
203
  const res = await this._http.fetchGet(`/api/loop/traces/${encodeURIComponent(traceId)}/candidates`);
160
204
  return res.candidates;
161
205
  }
206
+ // ── labels ────────────────────────────────────────────────────────────
207
+ /**
208
+ * Label a conversation, so it can be selected later.
209
+ *
210
+ * An average over everything is the least useful thing to train on. A model
211
+ * weak at refunds is fixed with refund examples, and you can only select
212
+ * those if you said so when the conversation arrived.
213
+ *
214
+ * A label may name a `parent`, which is what makes a MASTER GROUP: labelling
215
+ * something `refunds` with parent `billing` makes it selectable as either,
216
+ * and selecting `billing` later gathers every child without you maintaining
217
+ * a list of them.
218
+ *
219
+ * Labels are lowercased and trimmed, so `Refunds` and `refunds` are one label.
220
+ */
221
+ async label(traceId, labels, parent) {
222
+ const res = await this._http.fetchPost(`/api/loop/traces/${encodeURIComponent(traceId)}/labels`, { labels, parent });
223
+ return res.labels;
224
+ }
225
+ async unlabel(traceId, label) {
226
+ return this._http.fetchDelete(`/api/loop/traces/${encodeURIComponent(traceId)}/labels/${encodeURIComponent(label)}`);
227
+ }
228
+ /** Every label in the workspace, with how many conversations carry it. */
229
+ async listLabels() {
230
+ const res = await this._http.fetchGet('/api/loop/labels');
231
+ return res.labels;
232
+ }
162
233
  // ── training sets ─────────────────────────────────────────────────────
163
234
  /**
164
235
  * Build a training set from the feedback recorded so far.
package/dist/types.d.ts CHANGED
@@ -1960,7 +1960,15 @@ export type LoopMethod = 'sft' | 'dpo' | 'grpo';
1960
1960
  /** Who judged an answer. */
1961
1961
  export type LoopSource = 'human' | 'verifier' | 'judge' | 'behavioural';
1962
1962
  /** The judgement itself. */
1963
- export type LoopVerdict = 'accepted' | 'rejected' | 'edited' | 'scored';
1963
+ /**
1964
+ * The judgement recorded on an answer.
1965
+ *
1966
+ * `accepted` / `rejected` are the thumbs (see `upvote` / `downvote`).
1967
+ * `edited` says the model was WRONG and carries the better answer.
1968
+ * `gold` records the reference answer for the question, making no claim about
1969
+ * whether the model was right — which is why it is separate from `edited`.
1970
+ */
1971
+ export type LoopVerdict = 'accepted' | 'rejected' | 'edited' | 'scored' | 'gold';
1964
1972
  /** A function an assistant turn asked to invoke. `arguments` is JSON-encoded. */
1965
1973
  export interface LoopToolCall {
1966
1974
  id?: string;
@@ -2077,6 +2085,10 @@ export interface LoopDatasetCreateParams {
2077
2085
  deployment_id?: string;
2078
2086
  from?: string;
2079
2087
  to?: string;
2088
+ /** Narrow to one slice. A label with children selects them too. */
2089
+ label?: string;
2090
+ /** Take this many at random from what the filters matched. */
2091
+ sample?: number;
2080
2092
  /** Holds back your most recent work, not a random slice. 0-50. */
2081
2093
  holdout_percent?: number;
2082
2094
  max_items?: number;
@@ -2229,3 +2241,15 @@ export interface LoopGradeResult {
2229
2241
  signal_id: string | null;
2230
2242
  candidates_graded: number;
2231
2243
  }
2244
+ export interface LoopLabel {
2245
+ label: string;
2246
+ /** The master group this label belongs to, if any. */
2247
+ parent: string | null;
2248
+ created_by: string | null;
2249
+ created_at: string;
2250
+ }
2251
+ export interface LoopLabelCount {
2252
+ label: string;
2253
+ parent: string | null;
2254
+ traces: number;
2255
+ }
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "runbios-sdk",
3
- "version": "0.2.1-dev.107",
3
+ "version": "0.2.1-dev.109",
4
4
  "description": "Official TypeScript SDK for the Run BiOS training and deployment platform API",
5
5
  "type": "module",
6
6
  "main": "./dist/index.js",