runbios-sdk 0.2.1-dev.107 → 0.2.1-dev.109
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/dist/index.d.ts +1 -1
- package/dist/index.js +1 -1
- package/dist/resources/loop.d.ts +58 -1
- package/dist/resources/loop.js +71 -0
- package/dist/types.d.ts +25 -1
- package/package.json +1 -1
package/dist/index.d.ts
CHANGED
|
@@ -35,7 +35,7 @@ import { Loop } from './resources/loop.js';
|
|
|
35
35
|
* SDK version. Sent as part of the User-Agent header.
|
|
36
36
|
* Must match package.json "version" -- enforced by a contract test.
|
|
37
37
|
*/
|
|
38
|
-
export declare const VERSION = "0.2.1-dev.
|
|
38
|
+
export declare const VERSION = "0.2.1-dev.109";
|
|
39
39
|
export declare class RunBiOS {
|
|
40
40
|
/** Search models, fetch configs, check adapter compatibility. */
|
|
41
41
|
readonly models: Models;
|
package/dist/index.js
CHANGED
|
@@ -35,7 +35,7 @@ import { Loop } from './resources/loop.js';
|
|
|
35
35
|
* SDK version. Sent as part of the User-Agent header.
|
|
36
36
|
* Must match package.json "version" -- enforced by a contract test.
|
|
37
37
|
*/
|
|
38
|
-
export const VERSION = '0.2.1-dev.
|
|
38
|
+
export const VERSION = '0.2.1-dev.109';
|
|
39
39
|
export class RunBiOS {
|
|
40
40
|
/** Search models, fetch configs, check adapter compatibility. */
|
|
41
41
|
models;
|
package/dist/resources/loop.d.ts
CHANGED
|
@@ -1,5 +1,5 @@
|
|
|
1
1
|
import type { HttpClient } from '../client.js';
|
|
2
|
-
import type { LoopTrace, LoopTraceListResponse, LoopTraceListParams, LoopCaptureParams, LoopCaptureResult, LoopSignal, LoopSignalParams, LoopCandidate, LoopCandidateParams, LoopGrader, LoopGraderParams, LoopGradeResult, LoopDataset, LoopDatasetListParams, LoopDatasetItem, LoopDatasetCreateParams, LoopConfig, LoopConfigParams, LoopStats } from '../types.js';
|
|
2
|
+
import type { LoopTrace, LoopTraceListResponse, LoopTraceListParams, LoopCaptureParams, LoopCaptureResult, LoopSignal, LoopSignalParams, LoopCandidate, LoopCandidateParams, LoopGrader, LoopGraderParams, LoopLabel, LoopLabelCount, LoopGradeResult, LoopDataset, LoopDatasetListParams, LoopDatasetItem, LoopDatasetCreateParams, LoopConfig, LoopConfigParams, LoopStats } from '../types.js';
|
|
3
3
|
/**
|
|
4
4
|
* The Conscious Loop -- capture what your model was asked and answered, record
|
|
5
5
|
* whether it was right, and turn those judgements into training data.
|
|
@@ -88,6 +88,42 @@ export declare class Loop {
|
|
|
88
88
|
* behavioural hint.
|
|
89
89
|
*/
|
|
90
90
|
signal(traceId: string, params: LoopSignalParams): Promise<LoopSignal>;
|
|
91
|
+
/**
|
|
92
|
+
* Thumbs up: this answer was good. Shorthand for `signal(id, { verdict:
|
|
93
|
+
* 'accepted' })` — there is one stored vocabulary behind both, so a thumbs-up
|
|
94
|
+
* and an "accepted" can never disagree in your corpus.
|
|
95
|
+
*/
|
|
96
|
+
upvote(traceId: string, reason?: string): Promise<LoopSignal>;
|
|
97
|
+
/**
|
|
98
|
+
* Thumbs down: this answer was bad, and you have nothing better to offer.
|
|
99
|
+
*
|
|
100
|
+
* On its own this REMOVES the answer from training rather than teaching
|
|
101
|
+
* anything — there is no better version to learn. If you know what it should
|
|
102
|
+
* have said, `correct()` is worth far more.
|
|
103
|
+
*/
|
|
104
|
+
downvote(traceId: string, reason?: string): Promise<LoopSignal>;
|
|
105
|
+
/**
|
|
106
|
+
* The model was wrong; here is what it should have said.
|
|
107
|
+
*
|
|
108
|
+
* The most valuable feedback there is, because it produces BOTH halves of a
|
|
109
|
+
* preference pair from one action: the model learns your answer and learns to
|
|
110
|
+
* avoid its own.
|
|
111
|
+
*/
|
|
112
|
+
correct(traceId: string, answer: string, reason?: string): Promise<LoopSignal>;
|
|
113
|
+
/**
|
|
114
|
+
* Record the GOLD answer for this question — the reference, regardless of
|
|
115
|
+
* what the model happened to say.
|
|
116
|
+
*
|
|
117
|
+
* Different from `correct()` in a way that matters: a correction asserts the
|
|
118
|
+
* model was wrong, so the original becomes the rejected half of a pair. A gold
|
|
119
|
+
* answer asserts nothing about the model, so if it MATCHES what was said, no
|
|
120
|
+
* preference pair is invented — but SFT still learns it.
|
|
121
|
+
*
|
|
122
|
+
* It is the most reusable thing you can record: it trains SFT, forms a DPO
|
|
123
|
+
* pair when it differs from the answer, and stands in as the GRPO reference
|
|
124
|
+
* when you have not supplied a separate ground truth.
|
|
125
|
+
*/
|
|
126
|
+
gold(traceId: string, answer: string, reason?: string): Promise<LoopSignal>;
|
|
91
127
|
/** Every verdict on one conversation, oldest first. */
|
|
92
128
|
listSignals(traceId: string): Promise<LoopSignal[]>;
|
|
93
129
|
/**
|
|
@@ -121,6 +157,27 @@ export declare class Loop {
|
|
|
121
157
|
addCandidate(traceId: string, params: LoopCandidateParams): Promise<LoopCandidate>;
|
|
122
158
|
/** Every alternative answer recorded for one prompt, oldest first. */
|
|
123
159
|
listCandidates(traceId: string): Promise<LoopCandidate[]>;
|
|
160
|
+
/**
|
|
161
|
+
* Label a conversation, so it can be selected later.
|
|
162
|
+
*
|
|
163
|
+
* An average over everything is the least useful thing to train on. A model
|
|
164
|
+
* weak at refunds is fixed with refund examples, and you can only select
|
|
165
|
+
* those if you said so when the conversation arrived.
|
|
166
|
+
*
|
|
167
|
+
* A label may name a `parent`, which is what makes a MASTER GROUP: labelling
|
|
168
|
+
* something `refunds` with parent `billing` makes it selectable as either,
|
|
169
|
+
* and selecting `billing` later gathers every child without you maintaining
|
|
170
|
+
* a list of them.
|
|
171
|
+
*
|
|
172
|
+
* Labels are lowercased and trimmed, so `Refunds` and `refunds` are one label.
|
|
173
|
+
*/
|
|
174
|
+
label(traceId: string, labels: string[], parent?: string): Promise<LoopLabel[]>;
|
|
175
|
+
unlabel(traceId: string, label: string): Promise<{
|
|
176
|
+
deleted: boolean;
|
|
177
|
+
label: string;
|
|
178
|
+
}>;
|
|
179
|
+
/** Every label in the workspace, with how many conversations carry it. */
|
|
180
|
+
listLabels(): Promise<LoopLabelCount[]>;
|
|
124
181
|
/**
|
|
125
182
|
* Build a training set from the feedback recorded so far.
|
|
126
183
|
*
|
package/dist/resources/loop.js
CHANGED
|
@@ -116,6 +116,50 @@ export class Loop {
|
|
|
116
116
|
const res = await this._http.fetchPost(`/api/loop/traces/${encodeURIComponent(traceId)}/signals`, params);
|
|
117
117
|
return res.signal;
|
|
118
118
|
}
|
|
119
|
+
/**
|
|
120
|
+
* Thumbs up: this answer was good. Shorthand for `signal(id, { verdict:
|
|
121
|
+
* 'accepted' })` — there is one stored vocabulary behind both, so a thumbs-up
|
|
122
|
+
* and an "accepted" can never disagree in your corpus.
|
|
123
|
+
*/
|
|
124
|
+
async upvote(traceId, reason) {
|
|
125
|
+
return this.signal(traceId, { verdict: 'accepted', reason });
|
|
126
|
+
}
|
|
127
|
+
/**
|
|
128
|
+
* Thumbs down: this answer was bad, and you have nothing better to offer.
|
|
129
|
+
*
|
|
130
|
+
* On its own this REMOVES the answer from training rather than teaching
|
|
131
|
+
* anything — there is no better version to learn. If you know what it should
|
|
132
|
+
* have said, `correct()` is worth far more.
|
|
133
|
+
*/
|
|
134
|
+
async downvote(traceId, reason) {
|
|
135
|
+
return this.signal(traceId, { verdict: 'rejected', reason });
|
|
136
|
+
}
|
|
137
|
+
/**
|
|
138
|
+
* The model was wrong; here is what it should have said.
|
|
139
|
+
*
|
|
140
|
+
* The most valuable feedback there is, because it produces BOTH halves of a
|
|
141
|
+
* preference pair from one action: the model learns your answer and learns to
|
|
142
|
+
* avoid its own.
|
|
143
|
+
*/
|
|
144
|
+
async correct(traceId, answer, reason) {
|
|
145
|
+
return this.signal(traceId, { verdict: 'edited', correction: answer, reason });
|
|
146
|
+
}
|
|
147
|
+
/**
|
|
148
|
+
* Record the GOLD answer for this question — the reference, regardless of
|
|
149
|
+
* what the model happened to say.
|
|
150
|
+
*
|
|
151
|
+
* Different from `correct()` in a way that matters: a correction asserts the
|
|
152
|
+
* model was wrong, so the original becomes the rejected half of a pair. A gold
|
|
153
|
+
* answer asserts nothing about the model, so if it MATCHES what was said, no
|
|
154
|
+
* preference pair is invented — but SFT still learns it.
|
|
155
|
+
*
|
|
156
|
+
* It is the most reusable thing you can record: it trains SFT, forms a DPO
|
|
157
|
+
* pair when it differs from the answer, and stands in as the GRPO reference
|
|
158
|
+
* when you have not supplied a separate ground truth.
|
|
159
|
+
*/
|
|
160
|
+
async gold(traceId, answer, reason) {
|
|
161
|
+
return this.signal(traceId, { verdict: 'gold', correction: answer, reason });
|
|
162
|
+
}
|
|
119
163
|
/** Every verdict on one conversation, oldest first. */
|
|
120
164
|
async listSignals(traceId) {
|
|
121
165
|
const res = await this._http.fetchGet(`/api/loop/traces/${encodeURIComponent(traceId)}/signals`);
|
|
@@ -159,6 +203,33 @@ export class Loop {
|
|
|
159
203
|
const res = await this._http.fetchGet(`/api/loop/traces/${encodeURIComponent(traceId)}/candidates`);
|
|
160
204
|
return res.candidates;
|
|
161
205
|
}
|
|
206
|
+
// ── labels ────────────────────────────────────────────────────────────
|
|
207
|
+
/**
|
|
208
|
+
* Label a conversation, so it can be selected later.
|
|
209
|
+
*
|
|
210
|
+
* An average over everything is the least useful thing to train on. A model
|
|
211
|
+
* weak at refunds is fixed with refund examples, and you can only select
|
|
212
|
+
* those if you said so when the conversation arrived.
|
|
213
|
+
*
|
|
214
|
+
* A label may name a `parent`, which is what makes a MASTER GROUP: labelling
|
|
215
|
+
* something `refunds` with parent `billing` makes it selectable as either,
|
|
216
|
+
* and selecting `billing` later gathers every child without you maintaining
|
|
217
|
+
* a list of them.
|
|
218
|
+
*
|
|
219
|
+
* Labels are lowercased and trimmed, so `Refunds` and `refunds` are one label.
|
|
220
|
+
*/
|
|
221
|
+
async label(traceId, labels, parent) {
|
|
222
|
+
const res = await this._http.fetchPost(`/api/loop/traces/${encodeURIComponent(traceId)}/labels`, { labels, parent });
|
|
223
|
+
return res.labels;
|
|
224
|
+
}
|
|
225
|
+
async unlabel(traceId, label) {
|
|
226
|
+
return this._http.fetchDelete(`/api/loop/traces/${encodeURIComponent(traceId)}/labels/${encodeURIComponent(label)}`);
|
|
227
|
+
}
|
|
228
|
+
/** Every label in the workspace, with how many conversations carry it. */
|
|
229
|
+
async listLabels() {
|
|
230
|
+
const res = await this._http.fetchGet('/api/loop/labels');
|
|
231
|
+
return res.labels;
|
|
232
|
+
}
|
|
162
233
|
// ── training sets ─────────────────────────────────────────────────────
|
|
163
234
|
/**
|
|
164
235
|
* Build a training set from the feedback recorded so far.
|
package/dist/types.d.ts
CHANGED
|
@@ -1960,7 +1960,15 @@ export type LoopMethod = 'sft' | 'dpo' | 'grpo';
|
|
|
1960
1960
|
/** Who judged an answer. */
|
|
1961
1961
|
export type LoopSource = 'human' | 'verifier' | 'judge' | 'behavioural';
|
|
1962
1962
|
/** The judgement itself. */
|
|
1963
|
-
|
|
1963
|
+
/**
|
|
1964
|
+
* The judgement recorded on an answer.
|
|
1965
|
+
*
|
|
1966
|
+
* `accepted` / `rejected` are the thumbs (see `upvote` / `downvote`).
|
|
1967
|
+
* `edited` says the model was WRONG and carries the better answer.
|
|
1968
|
+
* `gold` records the reference answer for the question, making no claim about
|
|
1969
|
+
* whether the model was right — which is why it is separate from `edited`.
|
|
1970
|
+
*/
|
|
1971
|
+
export type LoopVerdict = 'accepted' | 'rejected' | 'edited' | 'scored' | 'gold';
|
|
1964
1972
|
/** A function an assistant turn asked to invoke. `arguments` is JSON-encoded. */
|
|
1965
1973
|
export interface LoopToolCall {
|
|
1966
1974
|
id?: string;
|
|
@@ -2077,6 +2085,10 @@ export interface LoopDatasetCreateParams {
|
|
|
2077
2085
|
deployment_id?: string;
|
|
2078
2086
|
from?: string;
|
|
2079
2087
|
to?: string;
|
|
2088
|
+
/** Narrow to one slice. A label with children selects them too. */
|
|
2089
|
+
label?: string;
|
|
2090
|
+
/** Take this many at random from what the filters matched. */
|
|
2091
|
+
sample?: number;
|
|
2080
2092
|
/** Holds back your most recent work, not a random slice. 0-50. */
|
|
2081
2093
|
holdout_percent?: number;
|
|
2082
2094
|
max_items?: number;
|
|
@@ -2229,3 +2241,15 @@ export interface LoopGradeResult {
|
|
|
2229
2241
|
signal_id: string | null;
|
|
2230
2242
|
candidates_graded: number;
|
|
2231
2243
|
}
|
|
2244
|
+
export interface LoopLabel {
|
|
2245
|
+
label: string;
|
|
2246
|
+
/** The master group this label belongs to, if any. */
|
|
2247
|
+
parent: string | null;
|
|
2248
|
+
created_by: string | null;
|
|
2249
|
+
created_at: string;
|
|
2250
|
+
}
|
|
2251
|
+
export interface LoopLabelCount {
|
|
2252
|
+
label: string;
|
|
2253
|
+
parent: string | null;
|
|
2254
|
+
traces: number;
|
|
2255
|
+
}
|